From 0a52d67d28449c442507c8e90f59d677f199094e Mon Sep 17 00:00:00 2001 From: Morten Hjorth-Jensen Date: Tue, 26 Aug 2025 07:02:47 +0200 Subject: [PATCH] addition --- doc/LectureNotes/week35.ipynb | 1599 +++++++++++++----- doc/pub/week35/html/._week35-bs000.html | 14 +- doc/pub/week35/html/._week35-bs001.html | 14 +- doc/pub/week35/html/._week35-bs002.html | 14 +- doc/pub/week35/html/._week35-bs003.html | 14 +- doc/pub/week35/html/._week35-bs004.html | 14 +- doc/pub/week35/html/._week35-bs005.html | 14 +- doc/pub/week35/html/._week35-bs006.html | 14 +- doc/pub/week35/html/._week35-bs007.html | 14 +- doc/pub/week35/html/._week35-bs008.html | 14 +- doc/pub/week35/html/._week35-bs009.html | 14 +- doc/pub/week35/html/._week35-bs010.html | 14 +- doc/pub/week35/html/._week35-bs011.html | 14 +- doc/pub/week35/html/._week35-bs012.html | 14 +- doc/pub/week35/html/._week35-bs013.html | 14 +- doc/pub/week35/html/._week35-bs014.html | 14 +- doc/pub/week35/html/._week35-bs015.html | 14 +- doc/pub/week35/html/._week35-bs016.html | 14 +- doc/pub/week35/html/._week35-bs017.html | 14 +- doc/pub/week35/html/._week35-bs018.html | 14 +- doc/pub/week35/html/._week35-bs019.html | 14 +- doc/pub/week35/html/._week35-bs020.html | 14 +- doc/pub/week35/html/._week35-bs021.html | 14 +- doc/pub/week35/html/._week35-bs022.html | 14 +- doc/pub/week35/html/._week35-bs023.html | 14 +- doc/pub/week35/html/._week35-bs024.html | 14 +- doc/pub/week35/html/._week35-bs025.html | 14 +- doc/pub/week35/html/._week35-bs026.html | 14 +- doc/pub/week35/html/._week35-bs027.html | 14 +- doc/pub/week35/html/._week35-bs028.html | 14 +- doc/pub/week35/html/._week35-bs029.html | 14 +- doc/pub/week35/html/._week35-bs030.html | 14 +- doc/pub/week35/html/._week35-bs031.html | 14 +- doc/pub/week35/html/._week35-bs032.html | 14 +- doc/pub/week35/html/._week35-bs033.html | 14 +- doc/pub/week35/html/._week35-bs034.html | 14 +- doc/pub/week35/html/._week35-bs035.html | 14 +- doc/pub/week35/html/._week35-bs036.html | 14 +- doc/pub/week35/html/._week35-bs037.html | 14 +- doc/pub/week35/html/._week35-bs038.html | 14 +- doc/pub/week35/html/._week35-bs039.html | 14 +- doc/pub/week35/html/._week35-bs040.html | 14 +- doc/pub/week35/html/._week35-bs041.html | 14 +- doc/pub/week35/html/._week35-bs042.html | 14 +- doc/pub/week35/html/._week35-bs043.html | 14 +- doc/pub/week35/html/._week35-bs044.html | 14 +- doc/pub/week35/html/._week35-bs045.html | 14 +- doc/pub/week35/html/._week35-bs046.html | 14 +- doc/pub/week35/html/._week35-bs047.html | 14 +- doc/pub/week35/html/._week35-bs048.html | 14 +- doc/pub/week35/html/._week35-bs049.html | 14 +- doc/pub/week35/html/._week35-bs050.html | 14 +- doc/pub/week35/html/._week35-bs051.html | 14 +- doc/pub/week35/html/._week35-bs052.html | 14 +- doc/pub/week35/html/._week35-bs053.html | 13 +- doc/pub/week35/html/._week35-bs054.html | 13 +- doc/pub/week35/html/._week35-bs055.html | 13 +- doc/pub/week35/html/._week35-bs056.html | 13 +- doc/pub/week35/html/._week35-bs057.html | 13 +- doc/pub/week35/html/._week35-bs058.html | 13 +- doc/pub/week35/html/._week35-bs059.html | 13 +- doc/pub/week35/html/._week35-bs060.html | 13 +- doc/pub/week35/html/._week35-bs061.html | 14 +- doc/pub/week35/html/._week35-bs062.html | 661 +++++++- doc/pub/week35/html/week35-bs.html | 14 +- doc/pub/week35/html/week35-reveal.html | 588 +++++++ doc/pub/week35/html/week35-solarized.html | 561 +++++- doc/pub/week35/html/week35.html | 561 +++++- doc/pub/week35/ipynb/ipynb-week35-src.tar.gz | Bin 191 -> 192 bytes doc/pub/week35/ipynb/week35.ipynb | 1587 +++++++++++++---- doc/src/week35/week35.do.txt | 502 ++++++ 71 files changed, 5999 insertions(+), 934 deletions(-) diff --git a/doc/LectureNotes/week35.ipynb b/doc/LectureNotes/week35.ipynb index 01ae64c94..b5da95d0a 100644 --- a/doc/LectureNotes/week35.ipynb +++ b/doc/LectureNotes/week35.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "f0622ebb", + "id": "d3666266", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "3fe98f20", + "id": "f9417425", "metadata": { "editable": true }, @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "8e604e3c", + "id": "aaba55da", "metadata": { "editable": true }, @@ -49,7 +49,7 @@ }, { "cell_type": "markdown", - "id": "c7f0b40a", + "id": "c4e2f992", "metadata": { "editable": true }, @@ -57,19 +57,21 @@ "### Reading recommendations:\n", "\n", "1. These lecture notes\n", - "\n", - "\n", "\n", - "2. Goodfellow, Bengio and Courville, Deep Learning, chapter 2 on linear algebra\n", + "2. Video of lecture at \n", "\n", - "3. Raschka et al on preprocessing of data, relevant for exercise 3 this week, see chapter 4.\n", + "3. Whiteboard notes at \n", "\n", - "4. For exercise 1 of week 35, the book by A. Aldo Faisal, Cheng Soon Ong, and Marc Peter Deisenroth on the Mathematics of Machine Learning, may be very relevant. In particular chapter 5 at URL\"https://mml-book.github.io/\" (section 5.5 on derivatives) is very useful for exercise 1 this coming week." + "4. Goodfellow, Bengio and Courville, Deep Learning, chapter 2 on linear algebra\n", + "\n", + "5. Raschka et al on preprocessing of data, relevant for exercise 3 this week, see chapter 4.\n", + "\n", + "6. For exercise 1 of week 35, the book by A. Aldo Faisal, Cheng Soon Ong, and Marc Peter Deisenroth on the Mathematics of Machine Learning, may be very relevant. In particular chapter 5 at URL\"https://mml-book.github.io/\" (section 5.5 on derivatives) is very useful for exercise 1 this coming week." ] }, { "cell_type": "markdown", - "id": "507709a4", + "id": "f0040522", "metadata": { "editable": true }, @@ -102,7 +104,7 @@ }, { "cell_type": "markdown", - "id": "bb3b5f67", + "id": "d64fcc78", "metadata": { "editable": true }, @@ -118,7 +120,7 @@ }, { "cell_type": "markdown", - "id": "14444be5", + "id": "a1ada6ce", "metadata": { "editable": true }, @@ -130,7 +132,7 @@ }, { "cell_type": "markdown", - "id": "4edc5258", + "id": "5645817b", "metadata": { "editable": true }, @@ -140,7 +142,7 @@ }, { "cell_type": "markdown", - "id": "4d6f9c43", + "id": "30bee0bb", "metadata": { "editable": true }, @@ -152,7 +154,7 @@ }, { "cell_type": "markdown", - "id": "fb52f7f5", + "id": "e6e20e11", "metadata": { "editable": true }, @@ -173,7 +175,7 @@ }, { "cell_type": "markdown", - "id": "4fff6b39", + "id": "7bfd72c1", "metadata": { "editable": true }, @@ -185,7 +187,7 @@ }, { "cell_type": "markdown", - "id": "54a4cabb", + "id": "69275ea0", "metadata": { "editable": true }, @@ -198,7 +200,7 @@ }, { "cell_type": "markdown", - "id": "d383a507", + "id": "976bb1cb", "metadata": { "editable": true }, @@ -210,7 +212,7 @@ }, { "cell_type": "markdown", - "id": "498fe31e", + "id": "0d85580a", "metadata": { "editable": true }, @@ -222,7 +224,7 @@ }, { "cell_type": "markdown", - "id": "7bbbdcd2", + "id": "5a8ac559", "metadata": { "editable": true }, @@ -232,7 +234,7 @@ }, { "cell_type": "markdown", - "id": "405a61ac", + "id": "64f74e34", "metadata": { "editable": true }, @@ -244,7 +246,7 @@ }, { "cell_type": "markdown", - "id": "e7c07f57", + "id": "8d503698", "metadata": { "editable": true }, @@ -257,7 +259,7 @@ }, { "cell_type": "markdown", - "id": "6054782f", + "id": "787c6923", "metadata": { "editable": true }, @@ -269,7 +271,7 @@ }, { "cell_type": "markdown", - "id": "7b55104f", + "id": "54c2b6dc", "metadata": { "editable": true }, @@ -279,7 +281,7 @@ }, { "cell_type": "markdown", - "id": "4635525e", + "id": "a8114ec8", "metadata": { "editable": true }, @@ -291,7 +293,7 @@ }, { "cell_type": "markdown", - "id": "11f6ffc7", + "id": "650e695e", "metadata": { "editable": true }, @@ -303,7 +305,7 @@ }, { "cell_type": "markdown", - "id": "820ddc3f", + "id": "885a3110", "metadata": { "editable": true }, @@ -314,7 +316,7 @@ }, { "cell_type": "markdown", - "id": "9aa4b864", + "id": "9fbaf50f", "metadata": { "editable": true }, @@ -326,7 +328,7 @@ }, { "cell_type": "markdown", - "id": "c542e149", + "id": "2376fa59", "metadata": { "editable": true }, @@ -345,7 +347,7 @@ }, { "cell_type": "markdown", - "id": "7be13194", + "id": "937df3c9", "metadata": { "editable": true }, @@ -358,7 +360,7 @@ }, { "cell_type": "markdown", - "id": "aad2641f", + "id": "e066a9bc", "metadata": { "editable": true }, @@ -368,7 +370,7 @@ }, { "cell_type": "markdown", - "id": "9dfe7e22", + "id": "af821847", "metadata": { "editable": true }, @@ -380,7 +382,7 @@ }, { "cell_type": "markdown", - "id": "852c82a9", + "id": "5e7ddc8b", "metadata": { "editable": true }, @@ -390,7 +392,7 @@ }, { "cell_type": "markdown", - "id": "84163c68", + "id": "47d5910e", "metadata": { "editable": true }, @@ -402,7 +404,7 @@ }, { "cell_type": "markdown", - "id": "5bfb8965", + "id": "6d62a75a", "metadata": { "editable": true }, @@ -412,7 +414,7 @@ }, { "cell_type": "markdown", - "id": "3822fa25", + "id": "58d10371", "metadata": { "editable": true }, @@ -424,7 +426,7 @@ }, { "cell_type": "markdown", - "id": "2c0f9f09", + "id": "1f349d3b", "metadata": { "editable": true }, @@ -435,7 +437,7 @@ }, { "cell_type": "markdown", - "id": "7ce9a08c", + "id": "14a30c64", "metadata": { "editable": true }, @@ -447,7 +449,7 @@ }, { "cell_type": "markdown", - "id": "3efb75ba", + "id": "b2faea7f", "metadata": { "editable": true }, @@ -457,7 +459,7 @@ }, { "cell_type": "markdown", - "id": "8c4ceb9d", + "id": "5810b814", "metadata": { "editable": true }, @@ -469,7 +471,7 @@ }, { "cell_type": "markdown", - "id": "53a5a827", + "id": "fcf3c26e", "metadata": { "editable": true }, @@ -479,7 +481,7 @@ }, { "cell_type": "markdown", - "id": "12d309d7", + "id": "a4453721", "metadata": { "editable": true }, @@ -491,7 +493,7 @@ }, { "cell_type": "markdown", - "id": "16fdbfbe", + "id": "38da802c", "metadata": { "editable": true }, @@ -511,7 +513,7 @@ }, { "cell_type": "markdown", - "id": "b5651e23", + "id": "bdc1020c", "metadata": { "editable": true }, @@ -538,7 +540,7 @@ }, { "cell_type": "markdown", - "id": "e3289213", + "id": "c69738c8", "metadata": { "editable": true }, @@ -550,7 +552,7 @@ }, { "cell_type": "markdown", - "id": "db9c8386", + "id": "c180c4b8", "metadata": { "editable": true }, @@ -562,7 +564,7 @@ }, { "cell_type": "markdown", - "id": "ea7d7d7e", + "id": "e273b465", "metadata": { "editable": true }, @@ -578,7 +580,7 @@ }, { "cell_type": "markdown", - "id": "7dace63c", + "id": "89031202", "metadata": { "editable": true }, @@ -596,7 +598,7 @@ }, { "cell_type": "markdown", - "id": "b00b4a82", + "id": "a5bfe37a", "metadata": { "editable": true }, @@ -608,7 +610,7 @@ }, { "cell_type": "markdown", - "id": "26392bc2", + "id": "9fe43da2", "metadata": { "editable": true }, @@ -620,7 +622,7 @@ }, { "cell_type": "markdown", - "id": "f5214ffc", + "id": "17f42b20", "metadata": { "editable": true }, @@ -631,7 +633,7 @@ }, { "cell_type": "markdown", - "id": "0eec52de", + "id": "6e12c604", "metadata": { "editable": true }, @@ -643,7 +645,7 @@ }, { "cell_type": "markdown", - "id": "077e4fe0", + "id": "c03524f3", "metadata": { "editable": true }, @@ -653,7 +655,7 @@ }, { "cell_type": "markdown", - "id": "e3ff215c", + "id": "66768181", "metadata": { "editable": true }, @@ -665,7 +667,7 @@ }, { "cell_type": "markdown", - "id": "6cb9ae59", + "id": "23864bbb", "metadata": { "editable": true }, @@ -679,7 +681,7 @@ }, { "cell_type": "markdown", - "id": "7a02062e", + "id": "c39fc9ec", "metadata": { "editable": true }, @@ -691,7 +693,7 @@ }, { "cell_type": "markdown", - "id": "45f73c0f", + "id": "9733262a", "metadata": { "editable": true }, @@ -703,7 +705,7 @@ }, { "cell_type": "markdown", - "id": "6c9d24a2", + "id": "dc160822", "metadata": { "editable": true }, @@ -715,7 +717,7 @@ }, { "cell_type": "markdown", - "id": "241fed59", + "id": "1211e28f", "metadata": { "editable": true }, @@ -725,7 +727,7 @@ }, { "cell_type": "markdown", - "id": "1c67de99", + "id": "939dfa18", "metadata": { "editable": true }, @@ -737,7 +739,7 @@ }, { "cell_type": "markdown", - "id": "36af470a", + "id": "f1d30438", "metadata": { "editable": true }, @@ -749,7 +751,7 @@ }, { "cell_type": "markdown", - "id": "38f9d9cf", + "id": "bdb64e7a", "metadata": { "editable": true }, @@ -761,7 +763,7 @@ }, { "cell_type": "markdown", - "id": "b20c89a1", + "id": "4952d52d", "metadata": { "editable": true }, @@ -775,7 +777,7 @@ }, { "cell_type": "markdown", - "id": "e56ba6c5", + "id": "f7ea9937", "metadata": { "editable": true }, @@ -787,7 +789,7 @@ }, { "cell_type": "markdown", - "id": "1481f969", + "id": "31e1ef19", "metadata": { "editable": true }, @@ -799,7 +801,7 @@ }, { "cell_type": "markdown", - "id": "fbd29e7c", + "id": "22628265", "metadata": { "editable": true }, @@ -811,7 +813,7 @@ }, { "cell_type": "markdown", - "id": "889c22f1", + "id": "d2e9c679", "metadata": { "editable": true }, @@ -821,7 +823,7 @@ }, { "cell_type": "markdown", - "id": "5a753c31", + "id": "c03929d2", "metadata": { "editable": true }, @@ -833,7 +835,7 @@ }, { "cell_type": "markdown", - "id": "4fed4676", + "id": "131c4d51", "metadata": { "editable": true }, @@ -843,7 +845,7 @@ }, { "cell_type": "markdown", - "id": "e1894902", + "id": "27cc18f9", "metadata": { "editable": true }, @@ -855,7 +857,7 @@ }, { "cell_type": "markdown", - "id": "ddca93b1", + "id": "f78fde44", "metadata": { "editable": true }, @@ -865,7 +867,7 @@ }, { "cell_type": "markdown", - "id": "5f9d8901", + "id": "b3117122", "metadata": { "editable": true }, @@ -877,7 +879,7 @@ }, { "cell_type": "markdown", - "id": "726c169c", + "id": "6025f5d9", "metadata": { "editable": true }, @@ -889,7 +891,7 @@ }, { "cell_type": "markdown", - "id": "7d7715d5", + "id": "f6d884e9", "metadata": { "editable": true }, @@ -901,7 +903,7 @@ }, { "cell_type": "markdown", - "id": "057fedd9", + "id": "d7241fbc", "metadata": { "editable": true }, @@ -916,7 +918,7 @@ }, { "cell_type": "markdown", - "id": "4c06fecd", + "id": "30ea6892", "metadata": { "editable": true }, @@ -928,7 +930,7 @@ }, { "cell_type": "markdown", - "id": "490e6469", + "id": "ada2be4f", "metadata": { "editable": true }, @@ -938,7 +940,7 @@ }, { "cell_type": "markdown", - "id": "b6259635", + "id": "131da9a2", "metadata": { "editable": true }, @@ -950,7 +952,7 @@ }, { "cell_type": "markdown", - "id": "0fda0191", + "id": "15d00d29", "metadata": { "editable": true }, @@ -960,7 +962,7 @@ }, { "cell_type": "markdown", - "id": "797e356f", + "id": "bd4077dd", "metadata": { "editable": true }, @@ -972,7 +974,7 @@ }, { "cell_type": "markdown", - "id": "46bdaee6", + "id": "a0fd8aac", "metadata": { "editable": true }, @@ -982,7 +984,7 @@ }, { "cell_type": "markdown", - "id": "cb1f5f36", + "id": "ae370d67", "metadata": { "editable": true }, @@ -994,7 +996,7 @@ }, { "cell_type": "markdown", - "id": "fa70c555", + "id": "687aa950", "metadata": { "editable": true }, @@ -1006,7 +1008,7 @@ }, { "cell_type": "markdown", - "id": "2fdb367f", + "id": "4159fec2", "metadata": { "editable": true }, @@ -1018,7 +1020,7 @@ }, { "cell_type": "markdown", - "id": "ced6532f", + "id": "e5f8e212", "metadata": { "editable": true }, @@ -1028,7 +1030,7 @@ }, { "cell_type": "markdown", - "id": "49c27708", + "id": "c26a9c8f", "metadata": { "editable": true }, @@ -1040,7 +1042,7 @@ }, { "cell_type": "markdown", - "id": "27da1de0", + "id": "859c590b", "metadata": { "editable": true }, @@ -1053,7 +1055,7 @@ }, { "cell_type": "markdown", - "id": "9ddef35d", + "id": "fbb8dd7f", "metadata": { "editable": true }, @@ -1065,7 +1067,7 @@ }, { "cell_type": "markdown", - "id": "ae006a7e", + "id": "bf096607", "metadata": { "editable": true }, @@ -1075,7 +1077,7 @@ }, { "cell_type": "markdown", - "id": "904d2486", + "id": "428e2635", "metadata": { "editable": true }, @@ -1087,7 +1089,7 @@ }, { "cell_type": "markdown", - "id": "525eda84", + "id": "81ff9c77", "metadata": { "editable": true }, @@ -1097,7 +1099,7 @@ }, { "cell_type": "markdown", - "id": "7eadfb46", + "id": "a4b9ca1c", "metadata": { "editable": true }, @@ -1109,7 +1111,7 @@ }, { "cell_type": "markdown", - "id": "8ee45bc9", + "id": "b64449f2", "metadata": { "editable": true }, @@ -1119,7 +1121,7 @@ }, { "cell_type": "markdown", - "id": "73b41248", + "id": "6a104348", "metadata": { "editable": true }, @@ -1131,7 +1133,7 @@ }, { "cell_type": "markdown", - "id": "dc4356aa", + "id": "ce07e978", "metadata": { "editable": true }, @@ -1141,7 +1143,7 @@ }, { "cell_type": "markdown", - "id": "3e2b94a9", + "id": "f81215cb", "metadata": { "editable": true }, @@ -1153,7 +1155,7 @@ }, { "cell_type": "markdown", - "id": "b9716f36", + "id": "54e70573", "metadata": { "editable": true }, @@ -1163,7 +1165,7 @@ }, { "cell_type": "markdown", - "id": "01341a5b", + "id": "2f2f537a", "metadata": { "editable": true }, @@ -1175,7 +1177,7 @@ }, { "cell_type": "markdown", - "id": "83a3f463", + "id": "9fb9df02", "metadata": { "editable": true }, @@ -1191,7 +1193,7 @@ }, { "cell_type": "markdown", - "id": "c2c42203", + "id": "2dff6039", "metadata": { "editable": true }, @@ -1203,7 +1205,7 @@ }, { "cell_type": "markdown", - "id": "94f060b9", + "id": "783debc8", "metadata": { "editable": true }, @@ -1213,7 +1215,7 @@ }, { "cell_type": "markdown", - "id": "d0e5c32c", + "id": "42965f7e", "metadata": { "editable": true }, @@ -1225,7 +1227,7 @@ }, { "cell_type": "markdown", - "id": "0b464ed4", + "id": "57e8d8fe", "metadata": { "editable": true }, @@ -1243,7 +1245,7 @@ }, { "cell_type": "markdown", - "id": "5c952119", + "id": "c43b98ff", "metadata": { "editable": true }, @@ -1255,7 +1257,7 @@ }, { "cell_type": "markdown", - "id": "25e4a5f3", + "id": "6ff581de", "metadata": { "editable": true }, @@ -1267,7 +1269,7 @@ }, { "cell_type": "markdown", - "id": "4f774737", + "id": "8f4462e2", "metadata": { "editable": true }, @@ -1277,7 +1279,7 @@ }, { "cell_type": "markdown", - "id": "59c91b82", + "id": "c935b169", "metadata": { "editable": true }, @@ -1289,7 +1291,7 @@ }, { "cell_type": "markdown", - "id": "33ceec0f", + "id": "50b87db5", "metadata": { "editable": true }, @@ -1299,7 +1301,7 @@ }, { "cell_type": "markdown", - "id": "931da8ef", + "id": "2599f182", "metadata": { "editable": true }, @@ -1311,7 +1313,7 @@ }, { "cell_type": "markdown", - "id": "2f2516b6", + "id": "6c07218b", "metadata": { "editable": true }, @@ -1321,7 +1323,7 @@ }, { "cell_type": "markdown", - "id": "bcb56a20", + "id": "d1058185", "metadata": { "editable": true }, @@ -1335,7 +1337,7 @@ }, { "cell_type": "markdown", - "id": "86333a6f", + "id": "951d1097", "metadata": { "editable": true }, @@ -1347,7 +1349,7 @@ }, { "cell_type": "markdown", - "id": "1513e73c", + "id": "73e7100b", "metadata": { "editable": true }, @@ -1358,7 +1360,7 @@ }, { "cell_type": "markdown", - "id": "c8c47d9a", + "id": "b7caf7f6", "metadata": { "editable": true }, @@ -1371,7 +1373,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "8ec18f31", + "id": "01229cfb", "metadata": { "collapsed": false, "editable": true @@ -1398,7 +1400,7 @@ }, { "cell_type": "markdown", - "id": "b7255330", + "id": "f4a3896e", "metadata": { "editable": true }, @@ -1409,7 +1411,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "9d5f3a70", + "id": "6ca9744d", "metadata": { "collapsed": false, "editable": true @@ -1422,7 +1424,7 @@ }, { "cell_type": "markdown", - "id": "8daa4681", + "id": "c434f72f", "metadata": { "editable": true }, @@ -1436,7 +1438,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "de91ca73", + "id": "d4131dbc", "metadata": { "collapsed": false, "editable": true @@ -1449,7 +1451,7 @@ }, { "cell_type": "markdown", - "id": "442ef680", + "id": "e5cff99e", "metadata": { "editable": true }, @@ -1460,7 +1462,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "9b329ab4", + "id": "c1b15227", "metadata": { "collapsed": false, "editable": true @@ -1472,7 +1474,7 @@ }, { "cell_type": "markdown", - "id": "d7ded522", + "id": "a0a432b4", "metadata": { "editable": true }, @@ -1483,7 +1485,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "34c18b8f", + "id": "319c1a33", "metadata": { "collapsed": false, "editable": true @@ -1499,7 +1501,7 @@ }, { "cell_type": "markdown", - "id": "229d94fa", + "id": "98d492e3", "metadata": { "editable": true }, @@ -1510,7 +1512,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "afb92eb9", + "id": "52ad17b3", "metadata": { "collapsed": false, "editable": true @@ -1524,7 +1526,7 @@ }, { "cell_type": "markdown", - "id": "8a02aaaf", + "id": "29abfbb7", "metadata": { "editable": true }, @@ -1545,7 +1547,7 @@ }, { "cell_type": "markdown", - "id": "b64b7d81", + "id": "503f0621", "metadata": { "editable": true }, @@ -1556,7 +1558,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "094eed3e", + "id": "6927de82", "metadata": { "collapsed": false, "editable": true @@ -1609,7 +1611,7 @@ }, { "cell_type": "markdown", - "id": "76119de5", + "id": "0b4f15f1", "metadata": { "editable": true }, @@ -1620,7 +1622,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "dff69c9a", + "id": "ac968b0f", "metadata": { "collapsed": false, "editable": true @@ -1645,7 +1647,7 @@ }, { "cell_type": "markdown", - "id": "02285c48", + "id": "4ea3a52b", "metadata": { "editable": true }, @@ -1657,7 +1659,7 @@ }, { "cell_type": "markdown", - "id": "71fab9ce", + "id": "a027a9a5", "metadata": { "editable": true }, @@ -1686,7 +1688,7 @@ }, { "cell_type": "markdown", - "id": "424914f7", + "id": "b0b3e66c", "metadata": { "editable": true }, @@ -1711,7 +1713,7 @@ }, { "cell_type": "markdown", - "id": "e486c793", + "id": "8af30d8b", "metadata": { "editable": true }, @@ -1731,7 +1733,7 @@ }, { "cell_type": "markdown", - "id": "1c6a9f34", + "id": "cd639711", "metadata": { "editable": true }, @@ -1758,7 +1760,7 @@ }, { "cell_type": "markdown", - "id": "28f4300e", + "id": "efb38ba0", "metadata": { "editable": true }, @@ -1771,7 +1773,7 @@ }, { "cell_type": "markdown", - "id": "deb48c69", + "id": "bcc3ee33", "metadata": { "editable": true }, @@ -1783,7 +1785,7 @@ }, { "cell_type": "markdown", - "id": "127f4812", + "id": "0c4c440b", "metadata": { "editable": true }, @@ -1794,7 +1796,7 @@ }, { "cell_type": "markdown", - "id": "305b4349", + "id": "b006a855", "metadata": { "editable": true }, @@ -1810,7 +1812,7 @@ { "cell_type": "code", "execution_count": 9, - "id": "9dacf721", + "id": "98079995", "metadata": { "collapsed": false, "editable": true @@ -1844,7 +1846,7 @@ }, { "cell_type": "markdown", - "id": "63596c47", + "id": "4d8bc84b", "metadata": { "editable": true }, @@ -1854,7 +1856,7 @@ }, { "cell_type": "markdown", - "id": "e663049c", + "id": "78d54efa", "metadata": { "editable": true }, @@ -1869,7 +1871,7 @@ }, { "cell_type": "markdown", - "id": "e2b36261", + "id": "2c6126fc", "metadata": { "editable": true }, @@ -1881,7 +1883,7 @@ }, { "cell_type": "markdown", - "id": "69fa292f", + "id": "2cc97a3e", "metadata": { "editable": true }, @@ -1891,7 +1893,7 @@ }, { "cell_type": "markdown", - "id": "23b3dece", + "id": "9e0ea279", "metadata": { "editable": true }, @@ -1907,7 +1909,7 @@ { "cell_type": "code", "execution_count": 10, - "id": "e1630ab8", + "id": "d00a25ea", "metadata": { "collapsed": false, "editable": true @@ -1924,7 +1926,7 @@ }, { "cell_type": "markdown", - "id": "c2d3b936", + "id": "864015a1", "metadata": { "editable": true }, @@ -1937,7 +1939,7 @@ { "cell_type": "code", "execution_count": 11, - "id": "355c6a66", + "id": "5ba30f30", "metadata": { "collapsed": false, "editable": true @@ -1984,7 +1986,7 @@ }, { "cell_type": "markdown", - "id": "ab553d39", + "id": "8dc3cb9f", "metadata": { "editable": true }, @@ -1998,7 +2000,7 @@ }, { "cell_type": "markdown", - "id": "f439180e", + "id": "068d5e0a", "metadata": { "editable": true }, @@ -2010,7 +2012,7 @@ }, { "cell_type": "markdown", - "id": "7ba7386d", + "id": "6feb2964", "metadata": { "editable": true }, @@ -2022,7 +2024,7 @@ }, { "cell_type": "markdown", - "id": "7e0a1440", + "id": "df3d9523", "metadata": { "editable": true }, @@ -2034,7 +2036,7 @@ }, { "cell_type": "markdown", - "id": "2e7611f3", + "id": "7e8785d1", "metadata": { "editable": true }, @@ -2044,7 +2046,7 @@ }, { "cell_type": "markdown", - "id": "414e53ca", + "id": "ef3be4f1", "metadata": { "editable": true }, @@ -2056,7 +2058,7 @@ }, { "cell_type": "markdown", - "id": "31ed8899", + "id": "3dcd8cf6", "metadata": { "editable": true }, @@ -2066,7 +2068,7 @@ }, { "cell_type": "markdown", - "id": "a66641fb", + "id": "7625f933", "metadata": { "editable": true }, @@ -2078,7 +2080,7 @@ }, { "cell_type": "markdown", - "id": "28a50f9e", + "id": "ef6f7710", "metadata": { "editable": true }, @@ -2089,7 +2091,7 @@ }, { "cell_type": "markdown", - "id": "b8c5f507", + "id": "b52129d8", "metadata": { "editable": true }, @@ -2101,7 +2103,7 @@ }, { "cell_type": "markdown", - "id": "445744c8", + "id": "c0670f24", "metadata": { "editable": true }, @@ -2113,7 +2115,7 @@ }, { "cell_type": "markdown", - "id": "ef90b7cb", + "id": "a30df157", "metadata": { "editable": true }, @@ -2123,7 +2125,7 @@ }, { "cell_type": "markdown", - "id": "9d4c2ac7", + "id": "75243d64", "metadata": { "editable": true }, @@ -2135,7 +2137,7 @@ }, { "cell_type": "markdown", - "id": "84e62cfc", + "id": "9c359540", "metadata": { "editable": true }, @@ -2147,7 +2149,7 @@ }, { "cell_type": "markdown", - "id": "41c2ba4b", + "id": "fa82ae4e", "metadata": { "editable": true }, @@ -2157,7 +2159,7 @@ }, { "cell_type": "markdown", - "id": "beb557ef", + "id": "65734ef1", "metadata": { "editable": true }, @@ -2169,7 +2171,7 @@ }, { "cell_type": "markdown", - "id": "b3fdffce", + "id": "650117d0", "metadata": { "editable": true }, @@ -2179,7 +2181,7 @@ }, { "cell_type": "markdown", - "id": "8005292a", + "id": "06771b2d", "metadata": { "editable": true }, @@ -2191,7 +2193,7 @@ }, { "cell_type": "markdown", - "id": "92d9f94b", + "id": "690e56c7", "metadata": { "editable": true }, @@ -2201,7 +2203,7 @@ }, { "cell_type": "markdown", - "id": "17940c71", + "id": "db5e7402", "metadata": { "editable": true }, @@ -2241,7 +2243,7 @@ }, { "cell_type": "markdown", - "id": "6ffb31dd", + "id": "a7813ad4", "metadata": { "editable": true }, @@ -2258,7 +2260,7 @@ }, { "cell_type": "markdown", - "id": "cc767251", + "id": "2108743e", "metadata": { "editable": true }, @@ -2281,7 +2283,7 @@ }, { "cell_type": "markdown", - "id": "347029b1", + "id": "3f63b5de", "metadata": { "editable": true }, @@ -2298,7 +2300,7 @@ }, { "cell_type": "markdown", - "id": "a77dda49", + "id": "05003c17", "metadata": { "editable": true }, @@ -2317,7 +2319,7 @@ }, { "cell_type": "markdown", - "id": "25a0d7d6", + "id": "315a72a0", "metadata": { "editable": true }, @@ -2328,7 +2330,7 @@ }, { "cell_type": "markdown", - "id": "70462fa0", + "id": "66ceac64", "metadata": { "editable": true }, @@ -2340,7 +2342,7 @@ }, { "cell_type": "markdown", - "id": "f8421c8e", + "id": "35b697b9", "metadata": { "editable": true }, @@ -2358,7 +2360,7 @@ }, { "cell_type": "markdown", - "id": "d4ef09ac", + "id": "4712a94c", "metadata": { "editable": true }, @@ -2374,7 +2376,7 @@ }, { "cell_type": "markdown", - "id": "fc6f8651", + "id": "c11c7415", "metadata": { "editable": true }, @@ -2386,7 +2388,7 @@ }, { "cell_type": "markdown", - "id": "ece72dfe", + "id": "7eecbc75", "metadata": { "editable": true }, @@ -2396,7 +2398,7 @@ }, { "cell_type": "markdown", - "id": "3f900d15", + "id": "3577f76a", "metadata": { "editable": true }, @@ -2409,7 +2411,7 @@ }, { "cell_type": "markdown", - "id": "3cb45c74", + "id": "e1cad910", "metadata": { "editable": true }, @@ -2421,7 +2423,7 @@ }, { "cell_type": "markdown", - "id": "81b8790f", + "id": "e1ed7f36", "metadata": { "editable": true }, @@ -2431,7 +2433,7 @@ }, { "cell_type": "markdown", - "id": "9d42c98f", + "id": "d9797b9a", "metadata": { "editable": true }, @@ -2444,7 +2446,7 @@ }, { "cell_type": "markdown", - "id": "ab036419", + "id": "8a7d60f5", "metadata": { "editable": true }, @@ -2454,7 +2456,7 @@ }, { "cell_type": "markdown", - "id": "bba11ff3", + "id": "597db6eb", "metadata": { "editable": true }, @@ -2466,7 +2468,7 @@ }, { "cell_type": "markdown", - "id": "6518184d", + "id": "03d92296", "metadata": { "editable": true }, @@ -2479,7 +2481,7 @@ }, { "cell_type": "markdown", - "id": "ba89e5b0", + "id": "135a3e67", "metadata": { "editable": true }, @@ -2492,7 +2494,7 @@ }, { "cell_type": "markdown", - "id": "47325cf5", + "id": "21a76d33", "metadata": { "editable": true }, @@ -2504,7 +2506,7 @@ }, { "cell_type": "markdown", - "id": "8d9036c1", + "id": "b7797c08", "metadata": { "editable": true }, @@ -2516,7 +2518,7 @@ }, { "cell_type": "markdown", - "id": "889f801b", + "id": "bd4d1e02", "metadata": { "editable": true }, @@ -2526,7 +2528,7 @@ }, { "cell_type": "markdown", - "id": "8a803d4a", + "id": "b46ffa29", "metadata": { "editable": true }, @@ -2539,7 +2541,7 @@ }, { "cell_type": "markdown", - "id": "ec92b840", + "id": "82955735", "metadata": { "editable": true }, @@ -2551,7 +2553,7 @@ }, { "cell_type": "markdown", - "id": "6de4ce40", + "id": "a489bfcb", "metadata": { "editable": true }, @@ -2563,7 +2565,7 @@ }, { "cell_type": "markdown", - "id": "22e97024", + "id": "b79dc723", "metadata": { "editable": true }, @@ -2575,7 +2577,7 @@ }, { "cell_type": "markdown", - "id": "9cd0a124", + "id": "b4e4b19c", "metadata": { "editable": true }, @@ -2587,7 +2589,7 @@ }, { "cell_type": "markdown", - "id": "29b31bed", + "id": "34dc0808", "metadata": { "editable": true }, @@ -2601,7 +2603,7 @@ }, { "cell_type": "markdown", - "id": "d1145ae7", + "id": "9e053bfc", "metadata": { "editable": true }, @@ -2613,7 +2615,7 @@ }, { "cell_type": "markdown", - "id": "f1479849", + "id": "b8cb20a9", "metadata": { "editable": true }, @@ -2623,7 +2625,7 @@ }, { "cell_type": "markdown", - "id": "0c51f5eb", + "id": "d218f1b3", "metadata": { "editable": true }, @@ -2635,7 +2637,7 @@ }, { "cell_type": "markdown", - "id": "2cbb0f42", + "id": "dc3e144e", "metadata": { "editable": true }, @@ -2647,7 +2649,7 @@ }, { "cell_type": "markdown", - "id": "406c4098", + "id": "b9dcf0b9", "metadata": { "editable": true }, @@ -2659,7 +2661,7 @@ }, { "cell_type": "markdown", - "id": "70966948", + "id": "a118bee6", "metadata": { "editable": true }, @@ -2671,7 +2673,7 @@ }, { "cell_type": "markdown", - "id": "627f3d38", + "id": "a18637c1", "metadata": { "editable": true }, @@ -2683,7 +2685,7 @@ }, { "cell_type": "markdown", - "id": "129105f7", + "id": "22ecf42d", "metadata": { "editable": true }, @@ -2706,7 +2708,7 @@ { "cell_type": "code", "execution_count": 12, - "id": "bce32748", + "id": "2c933b95", "metadata": { "collapsed": false, "editable": true @@ -2782,7 +2784,7 @@ }, { "cell_type": "markdown", - "id": "62eccb26", + "id": "27fcd2cd", "metadata": { "editable": true }, @@ -2794,7 +2796,7 @@ }, { "cell_type": "markdown", - "id": "675cc7de", + "id": "fb501ab3", "metadata": { "editable": true }, @@ -2809,7 +2811,7 @@ }, { "cell_type": "markdown", - "id": "ff36cd09", + "id": "bbdc8c03", "metadata": { "editable": true }, @@ -2821,7 +2823,7 @@ }, { "cell_type": "markdown", - "id": "7ae7a42a", + "id": "4f4e1b19", "metadata": { "editable": true }, @@ -2831,7 +2833,7 @@ }, { "cell_type": "markdown", - "id": "607325f0", + "id": "4b53a433", "metadata": { "editable": true }, @@ -2843,7 +2845,7 @@ }, { "cell_type": "markdown", - "id": "cfb328b4", + "id": "243e598d", "metadata": { "editable": true }, @@ -2853,7 +2855,7 @@ }, { "cell_type": "markdown", - "id": "bc44d844", + "id": "9e54e87c", "metadata": { "editable": true }, @@ -2865,7 +2867,7 @@ }, { "cell_type": "markdown", - "id": "17ddbfda", + "id": "ca9080f4", "metadata": { "editable": true }, @@ -2877,7 +2879,7 @@ }, { "cell_type": "markdown", - "id": "f5658751", + "id": "c0da749a", "metadata": { "editable": true }, @@ -2892,7 +2894,7 @@ }, { "cell_type": "markdown", - "id": "2b01e674", + "id": "73106e44", "metadata": { "editable": true }, @@ -2903,7 +2905,7 @@ }, { "cell_type": "markdown", - "id": "c18a65c6", + "id": "47a5df25", "metadata": { "editable": true }, @@ -2923,7 +2925,7 @@ }, { "cell_type": "markdown", - "id": "04258de2", + "id": "f75c5f9d", "metadata": { "editable": true }, @@ -2935,7 +2937,7 @@ }, { "cell_type": "markdown", - "id": "e10d0aad", + "id": "ae49bbbe", "metadata": { "editable": true }, @@ -2945,7 +2947,7 @@ }, { "cell_type": "markdown", - "id": "d198578b", + "id": "9927bcbf", "metadata": { "editable": true }, @@ -2957,7 +2959,7 @@ }, { "cell_type": "markdown", - "id": "2aec2fc1", + "id": "557aa14a", "metadata": { "editable": true }, @@ -2986,7 +2988,7 @@ }, { "cell_type": "markdown", - "id": "81fed6f5", + "id": "b77be05b", "metadata": { "editable": true }, @@ -3013,7 +3015,7 @@ }, { "cell_type": "markdown", - "id": "fe7d1d49", + "id": "dbca187b", "metadata": { "editable": true }, @@ -3024,7 +3026,7 @@ { "cell_type": "code", "execution_count": 13, - "id": "0b32041d", + "id": "52ee2c0a", "metadata": { "collapsed": false, "editable": true @@ -3064,7 +3066,7 @@ }, { "cell_type": "markdown", - "id": "db224b5b", + "id": "9585fb61", "metadata": { "editable": true }, @@ -3081,7 +3083,7 @@ }, { "cell_type": "markdown", - "id": "6b988f39", + "id": "1820219c", "metadata": { "editable": true }, @@ -3104,7 +3106,7 @@ }, { "cell_type": "markdown", - "id": "1214012e", + "id": "61e763b3", "metadata": { "editable": true }, @@ -3118,7 +3120,7 @@ }, { "cell_type": "markdown", - "id": "c7a8cd85", + "id": "d215271f", "metadata": { "editable": true }, @@ -3137,7 +3139,7 @@ }, { "cell_type": "markdown", - "id": "0f2a2211", + "id": "9e097ba9", "metadata": { "editable": true }, @@ -3147,7 +3149,7 @@ }, { "cell_type": "markdown", - "id": "e9863a0f", + "id": "c11e5ea0", "metadata": { "editable": true }, @@ -3159,7 +3161,7 @@ }, { "cell_type": "markdown", - "id": "d88c8316", + "id": "d7a4b260", "metadata": { "editable": true }, @@ -3173,7 +3175,7 @@ }, { "cell_type": "markdown", - "id": "643f1b6d", + "id": "dde8684e", "metadata": { "editable": true }, @@ -3185,7 +3187,7 @@ }, { "cell_type": "markdown", - "id": "0769ec85", + "id": "1ee4ec03", "metadata": { "editable": true }, @@ -3195,7 +3197,7 @@ }, { "cell_type": "markdown", - "id": "f0d61e66", + "id": "5db5fd8b", "metadata": { "editable": true }, @@ -3207,7 +3209,7 @@ }, { "cell_type": "markdown", - "id": "0d7c9d9e", + "id": "315fa545", "metadata": { "editable": true }, @@ -3224,7 +3226,7 @@ }, { "cell_type": "markdown", - "id": "335fb5ed", + "id": "669553b8", "metadata": { "editable": true }, @@ -3234,7 +3236,7 @@ }, { "cell_type": "markdown", - "id": "1dbb3e05", + "id": "a5c69189", "metadata": { "editable": true }, @@ -3250,7 +3252,7 @@ }, { "cell_type": "markdown", - "id": "4a535424", + "id": "c98988f9", "metadata": { "editable": true }, @@ -3260,7 +3262,7 @@ }, { "cell_type": "markdown", - "id": "512418d2", + "id": "ff274815", "metadata": { "editable": true }, @@ -3276,7 +3278,7 @@ }, { "cell_type": "markdown", - "id": "baa56321", + "id": "be481f90", "metadata": { "editable": true }, @@ -3286,7 +3288,7 @@ }, { "cell_type": "markdown", - "id": "1fd5881c", + "id": "5389a4fd", "metadata": { "editable": true }, @@ -3302,7 +3304,7 @@ }, { "cell_type": "markdown", - "id": "546a2bf3", + "id": "38c8b825", "metadata": { "editable": true }, @@ -3312,7 +3314,7 @@ }, { "cell_type": "markdown", - "id": "4e356417", + "id": "854c2801", "metadata": { "editable": true }, @@ -3329,7 +3331,7 @@ }, { "cell_type": "markdown", - "id": "1f3c787b", + "id": "77b24821", "metadata": { "editable": true }, @@ -3341,7 +3343,7 @@ }, { "cell_type": "markdown", - "id": "72018911", + "id": "9cb3e178", "metadata": { "editable": true }, @@ -3353,7 +3355,7 @@ }, { "cell_type": "markdown", - "id": "7d741e98", + "id": "a3e31623", "metadata": { "editable": true }, @@ -3365,7 +3367,7 @@ }, { "cell_type": "markdown", - "id": "f2e4a67f", + "id": "28150e6b", "metadata": { "editable": true }, @@ -3375,7 +3377,7 @@ }, { "cell_type": "markdown", - "id": "302c8edf", + "id": "fc1f871e", "metadata": { "editable": true }, @@ -3387,7 +3389,7 @@ }, { "cell_type": "markdown", - "id": "def3ec8a", + "id": "dd7f98b3", "metadata": { "editable": true }, @@ -3399,7 +3401,7 @@ }, { "cell_type": "markdown", - "id": "37150ee7", + "id": "b20a7cc6", "metadata": { "editable": true }, @@ -3411,7 +3413,7 @@ }, { "cell_type": "markdown", - "id": "e937a396", + "id": "82d661e0", "metadata": { "editable": true }, @@ -3421,7 +3423,7 @@ }, { "cell_type": "markdown", - "id": "07e76c7b", + "id": "705bb92a", "metadata": { "editable": true }, @@ -3433,7 +3435,7 @@ }, { "cell_type": "markdown", - "id": "99f80e22", + "id": "5c983fca", "metadata": { "editable": true }, @@ -3443,7 +3445,7 @@ }, { "cell_type": "markdown", - "id": "7a917bf2", + "id": "5b32695b", "metadata": { "editable": true }, @@ -3455,7 +3457,7 @@ }, { "cell_type": "markdown", - "id": "c11ae5fe", + "id": "c4285bb3", "metadata": { "editable": true }, @@ -3472,7 +3474,7 @@ }, { "cell_type": "markdown", - "id": "f300aa1b", + "id": "1fa01f9a", "metadata": { "editable": true }, @@ -3484,7 +3486,7 @@ }, { "cell_type": "markdown", - "id": "8bd1e2a6", + "id": "c9d412c2", "metadata": { "editable": true }, @@ -3496,7 +3498,7 @@ }, { "cell_type": "markdown", - "id": "62502292", + "id": "5307ebb8", "metadata": { "editable": true }, @@ -3506,7 +3508,7 @@ }, { "cell_type": "markdown", - "id": "efca6f6c", + "id": "ed2b38c3", "metadata": { "editable": true }, @@ -3518,7 +3520,7 @@ }, { "cell_type": "markdown", - "id": "e2cbd6f5", + "id": "5dd9e137", "metadata": { "editable": true }, @@ -3529,7 +3531,7 @@ }, { "cell_type": "markdown", - "id": "5cc9aea6", + "id": "38a3902d", "metadata": { "editable": true }, @@ -3541,7 +3543,7 @@ }, { "cell_type": "markdown", - "id": "61ad8f4c", + "id": "53a81f9f", "metadata": { "editable": true }, @@ -3551,7 +3553,7 @@ }, { "cell_type": "markdown", - "id": "b1cc9449", + "id": "c645a7d0", "metadata": { "editable": true }, @@ -3563,7 +3565,7 @@ }, { "cell_type": "markdown", - "id": "a9b28652", + "id": "a3098a03", "metadata": { "editable": true }, @@ -3573,7 +3575,7 @@ }, { "cell_type": "markdown", - "id": "f26b69cc", + "id": "08b90293", "metadata": { "editable": true }, @@ -3585,7 +3587,7 @@ }, { "cell_type": "markdown", - "id": "53acde0e", + "id": "1842a571", "metadata": { "editable": true }, @@ -3596,7 +3598,7 @@ }, { "cell_type": "markdown", - "id": "512b9f17", + "id": "0290fcb9", "metadata": { "editable": true }, @@ -3608,7 +3610,7 @@ }, { "cell_type": "markdown", - "id": "a1c49f5f", + "id": "ecbc00d0", "metadata": { "editable": true }, @@ -3626,7 +3628,7 @@ }, { "cell_type": "markdown", - "id": "db70d623", + "id": "670c8859", "metadata": { "editable": true }, @@ -3642,7 +3644,7 @@ }, { "cell_type": "markdown", - "id": "95b18b06", + "id": "286fd62d", "metadata": { "editable": true }, @@ -3654,7 +3656,7 @@ }, { "cell_type": "markdown", - "id": "db07fa1e", + "id": "a51c433e", "metadata": { "editable": true }, @@ -3666,7 +3668,7 @@ }, { "cell_type": "markdown", - "id": "9856cea3", + "id": "d19e4786", "metadata": { "editable": true }, @@ -3678,7 +3680,7 @@ }, { "cell_type": "markdown", - "id": "7f6cba4c", + "id": "56caedc2", "metadata": { "editable": true }, @@ -3691,7 +3693,7 @@ }, { "cell_type": "markdown", - "id": "581565a5", + "id": "d71c6ba0", "metadata": { "editable": true }, @@ -3707,7 +3709,7 @@ }, { "cell_type": "markdown", - "id": "22b63d08", + "id": "3dfd726a", "metadata": { "editable": true }, @@ -3721,7 +3723,7 @@ }, { "cell_type": "markdown", - "id": "cd6032aa", + "id": "5ed3d153", "metadata": { "editable": true }, @@ -3731,7 +3733,7 @@ }, { "cell_type": "markdown", - "id": "25b7936a", + "id": "1ac50487", "metadata": { "editable": true }, @@ -3743,7 +3745,7 @@ }, { "cell_type": "markdown", - "id": "9c4ad4d2", + "id": "d3cc0098", "metadata": { "editable": true }, @@ -3753,7 +3755,7 @@ }, { "cell_type": "markdown", - "id": "75bcfcea", + "id": "d1adc309", "metadata": { "editable": true }, @@ -3765,7 +3767,7 @@ }, { "cell_type": "markdown", - "id": "95c9cc40", + "id": "606b3621", "metadata": { "editable": true }, @@ -3775,7 +3777,7 @@ }, { "cell_type": "markdown", - "id": "33c19ed1", + "id": "48dfd7a6", "metadata": { "editable": true }, @@ -3789,7 +3791,7 @@ }, { "cell_type": "markdown", - "id": "c28b0a97", + "id": "f2d389fc", "metadata": { "editable": true }, @@ -3806,7 +3808,7 @@ }, { "cell_type": "markdown", - "id": "d4ece056", + "id": "93bee516", "metadata": { "editable": true }, @@ -3822,7 +3824,7 @@ }, { "cell_type": "markdown", - "id": "22360ee3", + "id": "b19050a0", "metadata": { "editable": true }, @@ -3834,7 +3836,7 @@ }, { "cell_type": "markdown", - "id": "77e8998d", + "id": "d6f57474", "metadata": { "editable": true }, @@ -3847,7 +3849,7 @@ }, { "cell_type": "markdown", - "id": "8ef4225d", + "id": "f7dd4e66", "metadata": { "editable": true }, @@ -3861,7 +3863,7 @@ }, { "cell_type": "markdown", - "id": "627eac2b", + "id": "99d565fb", "metadata": { "editable": true }, @@ -3871,7 +3873,7 @@ }, { "cell_type": "markdown", - "id": "f7ac0db1", + "id": "c65756ab", "metadata": { "editable": true }, @@ -3884,7 +3886,7 @@ }, { "cell_type": "markdown", - "id": "e93eae5d", + "id": "16c3efb7", "metadata": { "editable": true }, @@ -3903,7 +3905,7 @@ }, { "cell_type": "markdown", - "id": "58364673", + "id": "6e4194ff", "metadata": { "editable": true }, @@ -3915,7 +3917,7 @@ }, { "cell_type": "markdown", - "id": "30541096", + "id": "5259c2eb", "metadata": { "editable": true }, @@ -3927,7 +3929,7 @@ }, { "cell_type": "markdown", - "id": "101bb217", + "id": "1b927e23", "metadata": { "editable": true }, @@ -3937,7 +3939,7 @@ }, { "cell_type": "markdown", - "id": "ba8e52b1", + "id": "34ab019a", "metadata": { "editable": true }, @@ -3949,7 +3951,7 @@ }, { "cell_type": "markdown", - "id": "67d4335e", + "id": "21075d57", "metadata": { "editable": true }, @@ -3962,7 +3964,7 @@ }, { "cell_type": "markdown", - "id": "06416768", + "id": "8833f450", "metadata": { "editable": true }, @@ -3981,7 +3983,7 @@ }, { "cell_type": "markdown", - "id": "dd2f405a", + "id": "0813cc4c", "metadata": { "editable": true }, @@ -3991,7 +3993,7 @@ }, { "cell_type": "markdown", - "id": "40ded798", + "id": "50d7582d", "metadata": { "editable": true }, @@ -4010,7 +4012,7 @@ }, { "cell_type": "markdown", - "id": "d3044a01", + "id": "8f85e67d", "metadata": { "editable": true }, @@ -4028,7 +4030,7 @@ }, { "cell_type": "markdown", - "id": "d612ad97", + "id": "385c55d6", "metadata": { "editable": true }, @@ -4042,7 +4044,7 @@ }, { "cell_type": "markdown", - "id": "52f27c53", + "id": "ead461f7", "metadata": { "editable": true }, @@ -4057,7 +4059,7 @@ { "cell_type": "code", "execution_count": 14, - "id": "ca5ce77c", + "id": "80c333dc", "metadata": { "collapsed": false, "editable": true @@ -4078,7 +4080,7 @@ }, { "cell_type": "markdown", - "id": "1c02019d", + "id": "69f8b2df", "metadata": { "editable": true }, @@ -4095,7 +4097,7 @@ { "cell_type": "code", "execution_count": 15, - "id": "58381fbe", + "id": "4a07353d", "metadata": { "collapsed": false, "editable": true @@ -4127,7 +4129,7 @@ }, { "cell_type": "markdown", - "id": "a647044e", + "id": "ee3bce85", "metadata": { "editable": true }, @@ -4141,7 +4143,7 @@ }, { "cell_type": "markdown", - "id": "d7bea0a6", + "id": "3d15065a", "metadata": { "editable": true }, @@ -4154,7 +4156,7 @@ { "cell_type": "code", "execution_count": 16, - "id": "81d407da", + "id": "488f60ab", "metadata": { "collapsed": false, "editable": true @@ -4179,7 +4181,7 @@ }, { "cell_type": "markdown", - "id": "7998701a", + "id": "cafa6fa9", "metadata": { "editable": true }, @@ -4191,7 +4193,7 @@ }, { "cell_type": "markdown", - "id": "e4f65917", + "id": "ce70f3a8", "metadata": { "editable": true }, @@ -4203,7 +4205,7 @@ }, { "cell_type": "markdown", - "id": "73ecb476", + "id": "10c922ba", "metadata": { "editable": true }, @@ -4213,7 +4215,7 @@ }, { "cell_type": "markdown", - "id": "f4f10483", + "id": "0d791d47", "metadata": { "editable": true }, @@ -4230,7 +4232,7 @@ }, { "cell_type": "markdown", - "id": "9e173ba2", + "id": "533a270f", "metadata": { "editable": true }, @@ -4240,7 +4242,7 @@ }, { "cell_type": "markdown", - "id": "6b6f3d02", + "id": "1b8bd026", "metadata": { "editable": true }, @@ -4255,7 +4257,7 @@ }, { "cell_type": "markdown", - "id": "69a2c9b4", + "id": "f3f09455", "metadata": { "editable": true }, @@ -4265,7 +4267,7 @@ }, { "cell_type": "markdown", - "id": "41dedd73", + "id": "bce601da", "metadata": { "editable": true }, @@ -4279,7 +4281,7 @@ }, { "cell_type": "markdown", - "id": "a7b2bb8c", + "id": "85ed9781", "metadata": { "editable": true }, @@ -4291,7 +4293,7 @@ }, { "cell_type": "markdown", - "id": "dba35fc4", + "id": "7f6fcb9f", "metadata": { "editable": true }, @@ -4303,7 +4305,7 @@ }, { "cell_type": "markdown", - "id": "0e30a645", + "id": "6ea6f608", "metadata": { "editable": true }, @@ -4315,7 +4317,7 @@ }, { "cell_type": "markdown", - "id": "6fa482d3", + "id": "005927d3", "metadata": { "editable": true }, @@ -4325,7 +4327,7 @@ }, { "cell_type": "markdown", - "id": "ae9f30ae", + "id": "c1f7c3dd", "metadata": { "editable": true }, @@ -4337,7 +4339,7 @@ }, { "cell_type": "markdown", - "id": "67907a60", + "id": "a5c144b5", "metadata": { "editable": true }, @@ -4347,7 +4349,7 @@ }, { "cell_type": "markdown", - "id": "3a21fe71", + "id": "73802678", "metadata": { "editable": true }, @@ -4364,7 +4366,7 @@ }, { "cell_type": "markdown", - "id": "3b91b5d8", + "id": "aa888252", "metadata": { "editable": true }, @@ -4374,7 +4376,7 @@ }, { "cell_type": "markdown", - "id": "86de4409", + "id": "9d3363b0", "metadata": { "editable": true }, @@ -4386,7 +4388,7 @@ }, { "cell_type": "markdown", - "id": "d0c0608f", + "id": "49a1347a", "metadata": { "editable": true }, @@ -4396,7 +4398,7 @@ }, { "cell_type": "markdown", - "id": "98f544a6", + "id": "1ba2b422", "metadata": { "editable": true }, @@ -4408,7 +4410,7 @@ }, { "cell_type": "markdown", - "id": "24587410", + "id": "5ffae339", "metadata": { "editable": true }, @@ -4422,7 +4424,7 @@ }, { "cell_type": "markdown", - "id": "61fb4275", + "id": "cf729fa0", "metadata": { "editable": true }, @@ -4434,7 +4436,7 @@ }, { "cell_type": "markdown", - "id": "6d2404eb", + "id": "57cf0b92", "metadata": { "editable": true }, @@ -4456,7 +4458,7 @@ }, { "cell_type": "markdown", - "id": "22615b7e", + "id": "c61e0046", "metadata": { "editable": true }, @@ -4468,7 +4470,7 @@ }, { "cell_type": "markdown", - "id": "bee48c87", + "id": "66448cf4", "metadata": { "editable": true }, @@ -4483,7 +4485,7 @@ }, { "cell_type": "markdown", - "id": "7f12f53a", + "id": "53d03309", "metadata": { "editable": true }, @@ -4495,7 +4497,7 @@ }, { "cell_type": "markdown", - "id": "a6acb75c", + "id": "530e38d2", "metadata": { "editable": true }, @@ -4507,7 +4509,7 @@ }, { "cell_type": "markdown", - "id": "131e9d9c", + "id": "f073fbdc", "metadata": { "editable": true }, @@ -4517,7 +4519,7 @@ }, { "cell_type": "markdown", - "id": "85812d77", + "id": "cd18b54e", "metadata": { "editable": true }, @@ -4529,7 +4531,7 @@ }, { "cell_type": "markdown", - "id": "35007c74", + "id": "c21ebf43", "metadata": { "editable": true }, @@ -4539,7 +4541,7 @@ }, { "cell_type": "markdown", - "id": "0100935a", + "id": "7b868e3b", "metadata": { "editable": true }, @@ -4551,7 +4553,7 @@ }, { "cell_type": "markdown", - "id": "7b6ff86e", + "id": "e6e45cd5", "metadata": { "editable": true }, @@ -4561,7 +4563,7 @@ }, { "cell_type": "markdown", - "id": "1f452dfb", + "id": "d54cddec", "metadata": { "editable": true }, @@ -4573,7 +4575,7 @@ }, { "cell_type": "markdown", - "id": "05600fbd", + "id": "3f887335", "metadata": { "editable": true }, @@ -4590,7 +4592,7 @@ }, { "cell_type": "markdown", - "id": "a2af3e04", + "id": "43987fb6", "metadata": { "editable": true }, @@ -4603,7 +4605,7 @@ }, { "cell_type": "markdown", - "id": "0342c08c", + "id": "78741f5d", "metadata": { "editable": true }, @@ -4615,7 +4617,7 @@ }, { "cell_type": "markdown", - "id": "bdd5faeb", + "id": "ecf46f39", "metadata": { "editable": true }, @@ -4625,7 +4627,7 @@ }, { "cell_type": "markdown", - "id": "35873dc6", + "id": "f98d1ea2", "metadata": { "editable": true }, @@ -4638,7 +4640,7 @@ }, { "cell_type": "markdown", - "id": "6ff6ab78", + "id": "56a4e22a", "metadata": { "editable": true }, @@ -4648,7 +4650,7 @@ }, { "cell_type": "markdown", - "id": "77286b5f", + "id": "3087fc2c", "metadata": { "editable": true }, @@ -4660,7 +4662,7 @@ }, { "cell_type": "markdown", - "id": "0ade2616", + "id": "8d386a35", "metadata": { "editable": true }, @@ -4673,7 +4675,7 @@ }, { "cell_type": "markdown", - "id": "da22a168", + "id": "c8d58182", "metadata": { "editable": true }, @@ -4686,7 +4688,7 @@ }, { "cell_type": "markdown", - "id": "caa18da2", + "id": "15fe7dd6", "metadata": { "editable": true }, @@ -4698,7 +4700,7 @@ }, { "cell_type": "markdown", - "id": "f31c8084", + "id": "042bbee6", "metadata": { "editable": true }, @@ -4710,7 +4712,7 @@ }, { "cell_type": "markdown", - "id": "53dc71a2", + "id": "911f8e5c", "metadata": { "editable": true }, @@ -4720,7 +4722,7 @@ }, { "cell_type": "markdown", - "id": "753442d5", + "id": "40d4bbda", "metadata": { "editable": true }, @@ -4733,7 +4735,7 @@ }, { "cell_type": "markdown", - "id": "e0b2f58a", + "id": "8a00d027", "metadata": { "editable": true }, @@ -4745,7 +4747,7 @@ }, { "cell_type": "markdown", - "id": "e4af8e7e", + "id": "ff109317", "metadata": { "editable": true }, @@ -4757,7 +4759,7 @@ }, { "cell_type": "markdown", - "id": "6e353a3a", + "id": "4e5859eb", "metadata": { "editable": true }, @@ -4774,7 +4776,7 @@ }, { "cell_type": "markdown", - "id": "66a4056b", + "id": "75a8694b", "metadata": { "editable": true }, @@ -4786,7 +4788,7 @@ }, { "cell_type": "markdown", - "id": "ea5b3d9f", + "id": "80c47724", "metadata": { "editable": true }, @@ -4796,7 +4798,7 @@ }, { "cell_type": "markdown", - "id": "db50585a", + "id": "6b385376", "metadata": { "editable": true }, @@ -4808,7 +4810,7 @@ }, { "cell_type": "markdown", - "id": "afd3cd57", + "id": "32963935", "metadata": { "editable": true }, @@ -4818,7 +4820,7 @@ }, { "cell_type": "markdown", - "id": "6606774b", + "id": "de8c6756", "metadata": { "editable": true }, @@ -4830,7 +4832,7 @@ }, { "cell_type": "markdown", - "id": "bcfd3adb", + "id": "76b587d1", "metadata": { "editable": true }, @@ -4842,7 +4844,7 @@ }, { "cell_type": "markdown", - "id": "46d7227b", + "id": "8a1fc10c", "metadata": { "editable": true }, @@ -4858,7 +4860,7 @@ }, { "cell_type": "markdown", - "id": "a214b881", + "id": "a6377518", "metadata": { "editable": true }, @@ -4870,7 +4872,7 @@ }, { "cell_type": "markdown", - "id": "727c87de", + "id": "367d1837", "metadata": { "editable": true }, @@ -4882,7 +4884,7 @@ }, { "cell_type": "markdown", - "id": "5083858e", + "id": "2fa779fe", "metadata": { "editable": true }, @@ -4892,7 +4894,7 @@ }, { "cell_type": "markdown", - "id": "973ad2fa", + "id": "e0f99ac6", "metadata": { "editable": true }, @@ -4904,7 +4906,7 @@ }, { "cell_type": "markdown", - "id": "4fbd0028", + "id": "d2134909", "metadata": { "editable": true }, @@ -4914,7 +4916,7 @@ }, { "cell_type": "markdown", - "id": "bee26901", + "id": "09b0f22e", "metadata": { "editable": true }, @@ -4926,7 +4928,7 @@ }, { "cell_type": "markdown", - "id": "da771d96", + "id": "1d1987fc", "metadata": { "editable": true }, @@ -4943,7 +4945,7 @@ }, { "cell_type": "markdown", - "id": "8ecb8a62", + "id": "16f94ea4", "metadata": { "editable": true }, @@ -4955,7 +4957,7 @@ }, { "cell_type": "markdown", - "id": "5cf08e3d", + "id": "21c224fd", "metadata": { "editable": true }, @@ -4967,7 +4969,7 @@ }, { "cell_type": "markdown", - "id": "d984ce12", + "id": "80c9611e", "metadata": { "editable": true }, @@ -4977,7 +4979,7 @@ }, { "cell_type": "markdown", - "id": "837a69b7", + "id": "ca8706ca", "metadata": { "editable": true }, @@ -4989,7 +4991,7 @@ }, { "cell_type": "markdown", - "id": "21dde5ad", + "id": "6bcce614", "metadata": { "editable": true }, @@ -4999,7 +5001,7 @@ }, { "cell_type": "markdown", - "id": "4d49c112", + "id": "857b9ab1", "metadata": { "editable": true }, @@ -5011,7 +5013,7 @@ }, { "cell_type": "markdown", - "id": "5b4eabcf", + "id": "bb8c1546", "metadata": { "editable": true }, @@ -5021,7 +5023,7 @@ }, { "cell_type": "markdown", - "id": "28dcb3c8", + "id": "2f7636ba", "metadata": { "editable": true }, @@ -5033,7 +5035,7 @@ }, { "cell_type": "markdown", - "id": "f315a306", + "id": "07e308e9", "metadata": { "editable": true }, @@ -5043,7 +5045,7 @@ }, { "cell_type": "markdown", - "id": "7b4f990c", + "id": "7a20da6b", "metadata": { "editable": true }, @@ -5055,13 +5057,856 @@ }, { "cell_type": "markdown", - "id": "197daa81", + "id": "404a4814", "metadata": { "editable": true }, "source": [ "This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package [CVXOPT](https://cvxopt.org/). We will discuss how to code LASSO regression next week, when we have introduced gradient methods." ] + }, + { + "cell_type": "markdown", + "id": "c47881af", + "metadata": { + "editable": true + }, + "source": [ + "## Material for exercises week 35" + ] + }, + { + "cell_type": "markdown", + "id": "de625607", + "metadata": { + "editable": true + }, + "source": [ + "## Important technicalities: More on Rescaling data\n", + "\n", + "When you are comparing your own code with for example **Scikit-Learn**'s\n", + "library, there are some technicalities to keep in mind. The examples\n", + "here demonstrate some of these aspects with potential pitfalls.\n", + "\n", + "The discussion here focuses on the role of the intercept, how we can\n", + "set up the design matrix, what scaling we should use and other topics\n", + "which tend confuse us.\n", + "\n", + "The intercept can be interpreted as the expected value of our\n", + "target/output variables when all other predictors are set to zero.\n", + "Thus, if we cannot assume that the expected outputs/targets are zero\n", + "when all predictors are zero (the columns in the design matrix), it\n", + "may be a bad idea to implement a model which penalizes the intercept.\n", + "Furthermore, in for example Ridge and Lasso regression, the default solutions\n", + "from the library **Scikit-Learn** (when not shrinking $\\beta_0$) for the unknown parameters\n", + "$\\boldsymbol{\\beta}$, are derived under the assumption that both $\\boldsymbol{y}$ and\n", + "$\\boldsymbol{X}$ are zero centered, that is we subtract the mean values.\n", + "\n", + "If our predictors represent different scales, then it is important to\n", + "standardize the design matrix $\\boldsymbol{X}$ by subtracting the mean of each\n", + "column from the corresponding column and dividing the column with its\n", + "standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,\n", + "the results may differ. \n", + "\n", + "The\n", + "[Standardscaler](https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html)\n", + "function in **Scikit-Learn** does this for us. For the data sets we\n", + "have been studying in our various examples, the data are in many cases\n", + "already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a\n", + "survey of your data, with a critical assessment of them in case you need to scale the data.\n", + "\n", + "If you need to scale the data, not doing so will give an *unfair*\n", + "penalization of the parameters since their magnitude depends on the\n", + "scale of their corresponding predictor.\n", + "\n", + "The **Scikit-Learn** site has a good discussion of different ways of preprocessing data.\n", + "\n", + "Suppose as an example that you \n", + "you have an input variable given by the heights of different persons.\n", + "Human height might be measured in inches or meters or\n", + "kilometers. If measured in kilometers, a standard linear regression\n", + "model with this predictor would probably give a much bigger\n", + "coefficient term, than if measured in millimeters.\n", + "This can clearly lead to problems in evaluating the cost/loss functions.\n", + "\n", + "Keep in mind that when you transform your data set before training a model, the same transformation needs to be done\n", + "on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as" + ] + }, + { + "cell_type": "code", + "execution_count": 17, + "id": "c00a9805", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"\n", + "#Model training, we compute the mean value of y and X\n", + "y_train_mean = np.mean(y_train)\n", + "X_train_mean = np.mean(X_train,axis=0)\n", + "X_train = X_train - X_train_mean\n", + "y_train = y_train - y_train_mean\n", + "\n", + "# The we fit our model with the training data\n", + "trained_model = some_model.fit(X_train,y_train)\n", + "\n", + "\n", + "#Model prediction, we need also to transform our data set used for the prediction.\n", + "X_test = X_test - X_train_mean #Use mean from training data\n", + "y_pred = trained_model(X_test)\n", + "y_pred = y_pred + y_train_mean\n", + "\"\"\"" + ] + }, + { + "cell_type": "markdown", + "id": "b3fb9183", + "metadata": { + "editable": true + }, + "source": [ + "Let us try to understand what this may imply mathematically when we\n", + "subtract the mean values, also known as *zero centering*. For\n", + "simplicity, we will focus on ordinary regression, as done in the above example.\n", + "\n", + "The cost/loss function for regression is" + ] + }, + { + "cell_type": "markdown", + "id": "07a54955", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "C(\\beta_0, \\beta_1, ... , \\beta_{p-1}) = \\frac{1}{n}\\sum_{i=0}^{n} \\left(y_i - \\beta_0 - \\sum_{j=1}^{p-1} X_{ij}\\beta_j\\right)^2,.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1ac9c31d", + "metadata": { + "editable": true + }, + "source": [ + "Recall also that we use the squared value. This expression can lead to an\n", + "increased penalty for higher differences between predicted and\n", + "output/target values.\n", + "\n", + "What we have done is to single out the $\\beta_0$ term in the\n", + "definition of the mean squared error (MSE). The design matrix $X$\n", + "does in this case not contain any intercept column. When we take the\n", + "derivative with respect to $\\beta_0$, we want the derivative to obey" + ] + }, + { + "cell_type": "markdown", + "id": "ee6c6a4c", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial C}{\\partial \\beta_j} = 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "beb40c81", + "metadata": { + "editable": true + }, + "source": [ + "for all $j$. For $\\beta_0$ we have" + ] + }, + { + "cell_type": "markdown", + "id": "b051946e", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial C}{\\partial \\beta_0} = -\\frac{2}{n}\\sum_{i=0}^{n-1} \\left(y_i - \\beta_0 - \\sum_{j=1}^{p-1} X_{ij} \\beta_j\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "885a6b08", + "metadata": { + "editable": true + }, + "source": [ + "Multiplying away the constant $2/n$, we obtain" + ] + }, + { + "cell_type": "markdown", + "id": "5774d6be", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\sum_{i=0}^{n-1} \\beta_0 = \\sum_{i=0}^{n-1}y_i - \\sum_{i=0}^{n-1} \\sum_{j=1}^{p-1} X_{ij} \\beta_j.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "73ff6979", + "metadata": { + "editable": true + }, + "source": [ + "Let us specialize first to the case where we have only two parameters $\\beta_0$ and $\\beta_1$.\n", + "Our result for $\\beta_0$ simplifies then to" + ] + }, + { + "cell_type": "markdown", + "id": "792128bf", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "n\\beta_0 = \\sum_{i=0}^{n-1}y_i - \\sum_{i=0}^{n-1} X_{i1} \\beta_1.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "2be70ed5", + "metadata": { + "editable": true + }, + "source": [ + "We obtain then" + ] + }, + { + "cell_type": "markdown", + "id": "1d0c4ac1", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1}y_i - \\beta_1\\frac{1}{n}\\sum_{i=0}^{n-1} X_{i1}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "2d4a5936", + "metadata": { + "editable": true + }, + "source": [ + "If we define" + ] + }, + { + "cell_type": "markdown", + "id": "d30abc39", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\mu_{\\boldsymbol{x}_1}=\\frac{1}{n}\\sum_{i=0}^{n-1} X_{i1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "7f945f8b", + "metadata": { + "editable": true + }, + "source": [ + "and the mean value of the outputs as" + ] + }, + { + "cell_type": "markdown", + "id": "dff5bf57", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\mu_y=\\frac{1}{n}\\sum_{i=0}^{n-1}y_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "f222bcca", + "metadata": { + "editable": true + }, + "source": [ + "we have" + ] + }, + { + "cell_type": "markdown", + "id": "f1658ccd", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\mu_y - \\beta_1\\mu_{\\boldsymbol{x}_1}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "394ef3c4", + "metadata": { + "editable": true + }, + "source": [ + "In the general case with more parameters than $\\beta_0$ and $\\beta_1$, we have" + ] + }, + { + "cell_type": "markdown", + "id": "5f1dd1af", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1}y_i - \\frac{1}{n}\\sum_{i=0}^{n-1}\\sum_{j=1}^{p-1} X_{ij}\\beta_j.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0cafc2d3", + "metadata": { + "editable": true + }, + "source": [ + "We can rewrite the latter equation as" + ] + }, + { + "cell_type": "markdown", + "id": "2c878f98", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1}y_i - \\sum_{j=1}^{p-1} \\mu_{\\boldsymbol{x}_j}\\beta_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1dab41db", + "metadata": { + "editable": true + }, + "source": [ + "where we have defined" + ] + }, + { + "cell_type": "markdown", + "id": "cf2f2908", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\mu_{\\boldsymbol{x}_j}=\\frac{1}{n}\\sum_{i=0}^{n-1} X_{ij},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "20bac2a4", + "metadata": { + "editable": true + }, + "source": [ + "the mean value for all elements of the column vector $\\boldsymbol{x}_j$.\n", + "\n", + "Replacing $y_i$ with $y_i - y_i - \\overline{\\boldsymbol{y}}$ and centering also our design matrix results in a cost function (in vector-matrix disguise)" + ] + }, + { + "cell_type": "markdown", + "id": "0285afd3", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta}) = (\\boldsymbol{\\tilde{y}} - \\tilde{X}\\boldsymbol{\\beta})^T(\\boldsymbol{\\tilde{y}} - \\tilde{X}\\boldsymbol{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "ffff8823", + "metadata": { + "editable": true + }, + "source": [ + "If we minimize with respect to $\\boldsymbol{\\beta}$ we have then" + ] + }, + { + "cell_type": "markdown", + "id": "eea815a7", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\hat{\\boldsymbol{\\beta}} = (\\tilde{X}^T\\tilde{X})^{-1}\\tilde{X}^T\\boldsymbol{\\tilde{y}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0d935c60", + "metadata": { + "editable": true + }, + "source": [ + "where $\\boldsymbol{\\tilde{y}} = \\boldsymbol{y} - \\overline{\\boldsymbol{y}}$\n", + "and $\\tilde{X}_{ij} = X_{ij} - \\frac{1}{n}\\sum_{k=0}^{n-1}X_{kj}$.\n", + "\n", + "For Ridge regression we need to add $\\lambda \\boldsymbol{\\beta}^T\\boldsymbol{\\beta}$ to the cost function and get then" + ] + }, + { + "cell_type": "markdown", + "id": "bb4eabb8", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\hat{\\boldsymbol{\\beta}} = (\\tilde{X}^T\\tilde{X} + \\lambda I)^{-1}\\tilde{X}^T\\boldsymbol{\\tilde{y}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1da18d81", + "metadata": { + "editable": true + }, + "source": [ + "What does this mean? And why do we insist on all this? Let us look at some examples.\n", + "\n", + "This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*). Here our scaling of the data is done by subtracting the mean values only.\n", + "Note also that we do not split the data into training and test." + ] + }, + { + "cell_type": "code", + "execution_count": 18, + "id": "a11f0699", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "\n", + "np.random.seed(2021)\n", + "\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "\n", + "def fit_beta(X, y):\n", + " return np.linalg.pinv(X.T @ X) @ X.T @ y\n", + "\n", + "\n", + "true_beta = [2, 0.5, 3.7]\n", + "\n", + "x = np.linspace(0, 1, 11)\n", + "y = np.sum(\n", + " np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0\n", + ") + 0.1 * np.random.normal(size=len(x))\n", + "\n", + "degree = 3\n", + "X = np.zeros((len(x), degree))\n", + "\n", + "# Include the intercept in the design matrix\n", + "for p in range(degree):\n", + " X[:, p] = x ** p\n", + "\n", + "beta = fit_beta(X, y)\n", + "\n", + "# Intercept is included in the design matrix\n", + "skl = LinearRegression(fit_intercept=False).fit(X, y)\n", + "\n", + "print(f\"True beta: {true_beta}\")\n", + "print(f\"Fitted beta: {beta}\")\n", + "print(f\"Sklearn fitted beta: {skl.coef_}\")\n", + "ypredictOwn = X @ beta\n", + "ypredictSKL = skl.predict(X)\n", + "print(f\"MSE with intercept column\")\n", + "print(MSE(y,ypredictOwn))\n", + "print(f\"MSE with intercept column from SKL\")\n", + "print(MSE(y,ypredictSKL))\n", + "\n", + "\n", + "plt.figure()\n", + "plt.scatter(x, y, label=\"Data\")\n", + "plt.plot(x, X @ beta, label=\"Fit\")\n", + "plt.plot(x, skl.predict(X), label=\"Sklearn (fit_intercept=False)\")\n", + "\n", + "\n", + "# Do not include the intercept in the design matrix\n", + "X = np.zeros((len(x), degree - 1))\n", + "\n", + "for p in range(degree - 1):\n", + " X[:, p] = x ** (p + 1)\n", + "\n", + "# Intercept is not included in the design matrix\n", + "skl = LinearRegression(fit_intercept=True).fit(X, y)\n", + "\n", + "# Use centered values for X and y when computing coefficients\n", + "y_offset = np.average(y, axis=0)\n", + "X_offset = np.average(X, axis=0)\n", + "\n", + "beta = fit_beta(X - X_offset, y - y_offset)\n", + "intercept = np.mean(y_offset - X_offset @ beta)\n", + "\n", + "print(f\"Manual intercept: {intercept}\")\n", + "print(f\"Fitted beta (without intercept): {beta}\")\n", + "print(f\"Sklearn intercept: {skl.intercept_}\")\n", + "print(f\"Sklearn fitted beta (without intercept): {skl.coef_}\")\n", + "ypredictOwn = X @ beta\n", + "ypredictSKL = skl.predict(X)\n", + "print(f\"MSE with Manual intercept\")\n", + "print(MSE(y,ypredictOwn+intercept))\n", + "print(f\"MSE with Sklearn intercept\")\n", + "print(MSE(y,ypredictSKL))\n", + "\n", + "plt.plot(x, X @ beta + intercept, \"--\", label=\"Fit (manual intercept)\")\n", + "plt.plot(x, skl.predict(X), \"--\", label=\"Sklearn (fit_intercept=True)\")\n", + "plt.grid()\n", + "plt.legend()\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "ed61dd49", + "metadata": { + "editable": true + }, + "source": [ + "The intercept is the value of our output/target variable\n", + "when all our features are zero and our function crosses the $y$-axis (for a one-dimensional case). \n", + "\n", + "Printing the MSE, we see first that both methods give the same MSE, as\n", + "they should. However, when we move to for example Ridge regression,\n", + "the way we treat the intercept may give a larger or smaller MSE,\n", + "meaning that the MSE can be penalized by the value of the\n", + "intercept. Not including the intercept in the fit, means that the\n", + "regularization term does not include $\\beta_0$. For different values\n", + "of $\\lambda$, this may lead to different MSE values. \n", + "\n", + "To remind the reader, the regularization term, with the intercept in Ridge regression, is given by" + ] + }, + { + "cell_type": "markdown", + "id": "5de45190", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda \\vert\\vert \\boldsymbol{\\beta} \\vert\\vert_2^2 = \\lambda \\sum_{j=0}^{p-1}\\beta_j^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "6f53e8c1", + "metadata": { + "editable": true + }, + "source": [ + "but when we take out the intercept, this equation becomes" + ] + }, + { + "cell_type": "markdown", + "id": "3b0ffc74", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda \\vert\\vert \\boldsymbol{\\beta} \\vert\\vert_2^2 = \\lambda \\sum_{j=1}^{p-1}\\beta_j^2.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "10305919", + "metadata": { + "editable": true + }, + "source": [ + "For Lasso regression we have" + ] + }, + { + "cell_type": "markdown", + "id": "b2ce8814", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda \\vert\\vert \\boldsymbol{\\beta} \\vert\\vert_1 = \\lambda \\sum_{j=1}^{p-1}\\vert\\beta_j\\vert.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "e08c9fd7", + "metadata": { + "editable": true + }, + "source": [ + "It means that, when scaling the design matrix and the outputs/targets,\n", + "by subtracting the mean values, we have an optimization problem which\n", + "is not penalized by the intercept. The MSE value can then be smaller\n", + "since it focuses only on the remaining quantities. If we however bring\n", + "back the intercept, we will get a MSE which then contains the\n", + "intercept.\n", + "\n", + "Armed with this wisdom, we attempt first to simply set the intercept equal to **False** in our implementation of Ridge regression for our well-known vanilla data set." + ] + }, + { + "cell_type": "code", + "execution_count": 19, + "id": "5b7e1c63", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn import linear_model\n", + "\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "# Useful for eventual debugging.\n", + "np.random.seed(3155)\n", + "\n", + "n = 100\n", + "x = np.random.rand(n)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n", + "\n", + "Maxpolydegree = 20\n", + "X = np.zeros((n,Maxpolydegree))\n", + "#We include explicitely the intercept column\n", + "for degree in range(Maxpolydegree):\n", + " X[:,degree] = x**degree\n", + "# We split the data in test and training data\n", + "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n", + "\n", + "p = Maxpolydegree\n", + "I = np.eye(p,p)\n", + "# Decide which values of lambda to use\n", + "nlambdas = 6\n", + "MSEOwnRidgePredict = np.zeros(nlambdas)\n", + "MSERidgePredict = np.zeros(nlambdas)\n", + "lambdas = np.logspace(-4, 2, nlambdas)\n", + "for i in range(nlambdas):\n", + " lmb = lambdas[i]\n", + " OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train\n", + " # Note: we include the intercept column and no scaling\n", + " RegRidge = linear_model.Ridge(lmb,fit_intercept=False)\n", + " RegRidge.fit(X_train,y_train)\n", + " # and then make the prediction\n", + " ytildeOwnRidge = X_train @ OwnRidgeBeta\n", + " ypredictOwnRidge = X_test @ OwnRidgeBeta\n", + " ytildeRidge = RegRidge.predict(X_train)\n", + " ypredictRidge = RegRidge.predict(X_test)\n", + " MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n", + " MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n", + " print(\"Beta values for own Ridge implementation\")\n", + " print(OwnRidgeBeta)\n", + " print(\"Beta values for Scikit-Learn Ridge implementation\")\n", + " print(RegRidge.coef_)\n", + " print(\"MSE values for own Ridge implementation\")\n", + " print(MSEOwnRidgePredict[i])\n", + " print(\"MSE values for Scikit-Learn Ridge implementation\")\n", + " print(MSERidgePredict[i])\n", + "\n", + "# Now plot the results\n", + "plt.figure()\n", + "plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')\n", + "plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')\n", + "\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('MSE')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "bbf9639e", + "metadata": { + "editable": true + }, + "source": [ + "The results here agree when we force **Scikit-Learn**'s Ridge function to include the first column in our design matrix.\n", + "We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.\n", + "What happens if we do not include the intercept in our fit?\n", + "Let us see how we can change this code by zero centering." + ] + }, + { + "cell_type": "code", + "execution_count": 20, + "id": "3a82ee91", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn import linear_model\n", + "from sklearn.preprocessing import StandardScaler\n", + "\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "# Useful for eventual debugging.\n", + "np.random.seed(315)\n", + "\n", + "n = 100\n", + "x = np.random.rand(n)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n", + "\n", + "Maxpolydegree = 20\n", + "X = np.zeros((n,Maxpolydegree-1))\n", + "\n", + "for degree in range(1,Maxpolydegree): #No intercept column\n", + " X[:,degree-1] = x**(degree)\n", + "\n", + "# We split the data in test and training data\n", + "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n", + "\n", + "#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable\n", + "X_train_mean = np.mean(X_train,axis=0)\n", + "#Center by removing mean from each feature\n", + "X_train_scaled = X_train - X_train_mean \n", + "X_test_scaled = X_test - X_train_mean\n", + "#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)\n", + "#Remove the intercept from the training data.\n", + "y_scaler = np.mean(y_train) \n", + "y_train_scaled = y_train - y_scaler \n", + "\n", + "p = Maxpolydegree-1\n", + "I = np.eye(p,p)\n", + "# Decide which values of lambda to use\n", + "nlambdas = 6\n", + "MSEOwnRidgePredict = np.zeros(nlambdas)\n", + "MSERidgePredict = np.zeros(nlambdas)\n", + "\n", + "lambdas = np.logspace(-4, 2, nlambdas)\n", + "for i in range(nlambdas):\n", + " lmb = lambdas[i]\n", + " OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)\n", + " intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data\n", + " #Add intercept to prediction\n", + " ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler \n", + " RegRidge = linear_model.Ridge(lmb)\n", + " RegRidge.fit(X_train,y_train)\n", + " ypredictRidge = RegRidge.predict(X_test)\n", + " MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n", + " MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n", + " print(\"Beta values for own Ridge implementation\")\n", + " print(OwnRidgeBeta) #Intercept is given by mean of target variable\n", + " print(\"Beta values for Scikit-Learn Ridge implementation\")\n", + " print(RegRidge.coef_)\n", + " print('Intercept from own implementation:')\n", + " print(intercept_)\n", + " print('Intercept from Scikit-Learn Ridge implementation')\n", + " print(RegRidge.intercept_)\n", + " print(\"MSE values for own Ridge implementation\")\n", + " print(MSEOwnRidgePredict[i])\n", + " print(\"MSE values for Scikit-Learn Ridge implementation\")\n", + " print(MSERidgePredict[i])\n", + "\n", + "\n", + "# Now plot the results\n", + "plt.figure()\n", + "plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')\n", + "plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('MSE')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "c061da66", + "metadata": { + "editable": true + }, + "source": [ + "We see here, when compared to the code which includes explicitely the\n", + "intercept column, that our MSE value is actually smaller. This is\n", + "because the regularization term does not include the intercept value\n", + "$\\beta_0$ in the fitting. This applies to Lasso regularization as\n", + "well. It means that our optimization is now done only with the\n", + "centered matrix and/or vector that enter the fitting procedure." + ] } ], "metadata": {}, diff --git a/doc/pub/week35/html/._week35-bs000.html b/doc/pub/week35/html/._week35-bs000.html index 028edad4c..ad2ecd4da 100644 --- a/doc/pub/week35/html/._week35-bs000.html +++ b/doc/pub/week35/html/._week35-bs000.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -359,7 +369,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs001.html b/doc/pub/week35/html/._week35-bs001.html index c593ee764..ce8945f28 100644 --- a/doc/pub/week35/html/._week35-bs001.html +++ b/doc/pub/week35/html/._week35-bs001.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -356,7 +366,7 @@ MathJax.Hub.Config({
  • 10
  • 11
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs002.html b/doc/pub/week35/html/._week35-bs002.html index 1bb1245b2..96512ad35 100644 --- a/doc/pub/week35/html/._week35-bs002.html +++ b/doc/pub/week35/html/._week35-bs002.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -355,7 +365,7 @@ Similarly, Mehta et al
  • 11
  • 12
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs003.html b/doc/pub/week35/html/._week35-bs003.html index 1a2bcf6bc..3097db68c 100644 --- a/doc/pub/week35/html/._week35-bs003.html +++ b/doc/pub/week35/html/._week35-bs003.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -380,7 +390,7 @@ values \( \tilde{y}_i \), namely the so-called cost/loss function.
  • 12
  • 13
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs004.html b/doc/pub/week35/html/._week35-bs004.html index 0e606ba5d..65624f41d 100644 --- a/doc/pub/week35/html/._week35-bs004.html +++ b/doc/pub/week35/html/._week35-bs004.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -362,7 +372,7 @@ $$
  • 13
  • 14
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs005.html b/doc/pub/week35/html/._week35-bs005.html index 6d5c26e46..871701572 100644 --- a/doc/pub/week35/html/._week35-bs005.html +++ b/doc/pub/week35/html/._week35-bs005.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -385,7 +395,7 @@ $$
  • 14
  • 15
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs006.html b/doc/pub/week35/html/._week35-bs006.html index ddc19a43a..4c1776c47 100644 --- a/doc/pub/week35/html/._week35-bs006.html +++ b/doc/pub/week35/html/._week35-bs006.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -381,7 +391,7 @@ allow for the usage of direct linear algebra methods such as LU decomposi
  • 15
  • 16
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs007.html b/doc/pub/week35/html/._week35-bs007.html index fa29b1fd7..9b3c8d2a8 100644 --- a/doc/pub/week35/html/._week35-bs007.html +++ b/doc/pub/week35/html/._week35-bs007.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -368,7 +378,7 @@ $$
  • 16
  • 17
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs008.html b/doc/pub/week35/html/._week35-bs008.html index f646e30f9..5829d3cec 100644 --- a/doc/pub/week35/html/._week35-bs008.html +++ b/doc/pub/week35/html/._week35-bs008.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -366,7 +376,7 @@ vector is differentiable.
  • 17
  • 18
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs009.html b/doc/pub/week35/html/._week35-bs009.html index 1a682b3b9..9b0988ca2 100644 --- a/doc/pub/week35/html/._week35-bs009.html +++ b/doc/pub/week35/html/._week35-bs009.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -365,7 +375,7 @@ $$
  • 18
  • 19
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs010.html b/doc/pub/week35/html/._week35-bs010.html index 5907bd1ef..7da0f958b 100644 --- a/doc/pub/week35/html/._week35-bs010.html +++ b/doc/pub/week35/html/._week35-bs010.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -376,7 +386,7 @@ $$
  • 19
  • 20
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs011.html b/doc/pub/week35/html/._week35-bs011.html index 17b5510cf..3a09a7a2f 100644 --- a/doc/pub/week35/html/._week35-bs011.html +++ b/doc/pub/week35/html/._week35-bs011.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -378,7 +388,7 @@ $$
  • 20
  • 21
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs012.html b/doc/pub/week35/html/._week35-bs012.html index a9ab18ebf..14877577f 100644 --- a/doc/pub/week35/html/._week35-bs012.html +++ b/doc/pub/week35/html/._week35-bs012.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -379,7 +389,7 @@ $$
  • 21
  • 22
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs013.html b/doc/pub/week35/html/._week35-bs013.html index 341b965f5..b863a98f4 100644 --- a/doc/pub/week35/html/._week35-bs013.html +++ b/doc/pub/week35/html/._week35-bs013.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -391,7 +401,7 @@ $$
  • 22
  • 23
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs014.html b/doc/pub/week35/html/._week35-bs014.html index eef1a405d..716b6dccd 100644 --- a/doc/pub/week35/html/._week35-bs014.html +++ b/doc/pub/week35/html/._week35-bs014.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -374,7 +384,7 @@ problem. We will discuss this in greater detail next week when we introduce grad
  • 23
  • 24
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs015.html b/doc/pub/week35/html/._week35-bs015.html index 3287b99ca..1ca3848ef 100644 --- a/doc/pub/week35/html/._week35-bs015.html +++ b/doc/pub/week35/html/._week35-bs015.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -369,7 +379,7 @@ $$
  • 24
  • 25
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs016.html b/doc/pub/week35/html/._week35-bs016.html index 8a910a263..7720f1063 100644 --- a/doc/pub/week35/html/._week35-bs016.html +++ b/doc/pub/week35/html/._week35-bs016.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -358,7 +368,7 @@ $$
  • 25
  • 26
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs017.html b/doc/pub/week35/html/._week35-bs017.html index 148f1002d..173b00f24 100644 --- a/doc/pub/week35/html/._week35-bs017.html +++ b/doc/pub/week35/html/._week35-bs017.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -411,7 +421,7 @@ ytildenp = np.<
  • 26
  • 27
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs018.html b/doc/pub/week35/html/._week35-bs018.html index 1c4a55cf9..f0acf50d5 100644 --- a/doc/pub/week35/html/._week35-bs018.html +++ b/doc/pub/week35/html/._week35-bs018.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -452,7 +462,7 @@ Since we are not using Scikit-Learn here we can define our own \( R2 \) f
  • 27
  • 28
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs019.html b/doc/pub/week35/html/._week35-bs019.html index dc5296e60..23ea32c35 100644 --- a/doc/pub/week35/html/._week35-bs019.html +++ b/doc/pub/week35/html/._week35-bs019.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -365,7 +375,7 @@ but now splitting the data into a training set and a test set.
  • 28
  • 29
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs020.html b/doc/pub/week35/html/._week35-bs020.html index ab71f10d5..5ce86a863 100644 --- a/doc/pub/week35/html/._week35-bs020.html +++ b/doc/pub/week35/html/._week35-bs020.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -409,7 +419,7 @@ ypredict = X_test 29
  • 30
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs021.html b/doc/pub/week35/html/._week35-bs021.html index 8b3117821..fbd6f7d25 100644 --- a/doc/pub/week35/html/._week35-bs021.html +++ b/doc/pub/week35/html/._week35-bs021.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -387,7 +397,7 @@ normally recommend using the latter functionality.
  • 30
  • 31
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs022.html b/doc/pub/week35/html/._week35-bs022.html index 29628a345..83c5781d3 100644 --- a/doc/pub/week35/html/._week35-bs022.html +++ b/doc/pub/week35/html/._week35-bs022.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -374,7 +384,7 @@ visualization.
  • 31
  • 32
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs023.html b/doc/pub/week35/html/._week35-bs023.html index 02fa39390..acf3c7872 100644 --- a/doc/pub/week35/html/._week35-bs023.html +++ b/doc/pub/week35/html/._week35-bs023.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -369,7 +379,7 @@ the features in a way to avoid such outlier values.
  • 32
  • 33
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs024.html b/doc/pub/week35/html/._week35-bs024.html index a6e7d3b19..fbaa2866a 100644 --- a/doc/pub/week35/html/._week35-bs024.html +++ b/doc/pub/week35/html/._week35-bs024.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -357,7 +367,7 @@ ensures that all features are exactly between \( 0 \) and \( 1 \). The
  • 33
  • 34
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs025.html b/doc/pub/week35/html/._week35-bs025.html index 25ff389f5..721e96915 100644 --- a/doc/pub/week35/html/._week35-bs025.html +++ b/doc/pub/week35/html/._week35-bs025.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -371,7 +381,7 @@ techniques.
  • 34
  • 35
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs026.html b/doc/pub/week35/html/._week35-bs026.html index 3f1ab1c3f..5102d3bb0 100644 --- a/doc/pub/week35/html/._week35-bs026.html +++ b/doc/pub/week35/html/._week35-bs026.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -358,7 +368,7 @@ This ensures that each feature has zero mean and unit standard deviation. For d
  • 35
  • 36
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs027.html b/doc/pub/week35/html/._week35-bs027.html index 71fcaad39..dbba0ebb4 100644 --- a/doc/pub/week35/html/._week35-bs027.html +++ b/doc/pub/week35/html/._week35-bs027.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -399,7 +409,7 @@ display(XPandas-Xscaled)
  • 36
  • 37
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs028.html b/doc/pub/week35/html/._week35-bs028.html index 1c6aedfce..f5ea48e77 100644 --- a/doc/pub/week35/html/._week35-bs028.html +++ b/doc/pub/week35/html/._week35-bs028.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -358,7 +368,7 @@ $$
  • 37
  • 38
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs029.html b/doc/pub/week35/html/._week35-bs029.html index 761899986..fe97ec5fa 100644 --- a/doc/pub/week35/html/._week35-bs029.html +++ b/doc/pub/week35/html/._week35-bs029.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -441,7 +451,7 @@ plt.show()
  • 38
  • 39
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs030.html b/doc/pub/week35/html/._week35-bs030.html index 1e4b74a19..26430cd6d 100644 --- a/doc/pub/week35/html/._week35-bs030.html +++ b/doc/pub/week35/html/._week35-bs030.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -376,7 +386,7 @@ We can then interpret our optimal model \( \tilde{\boldsymbol{y}} \) as being re
  • 39
  • 40
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs031.html b/doc/pub/week35/html/._week35-bs031.html index 9adcb5f29..8c3685346 100644 --- a/doc/pub/week35/html/._week35-bs031.html +++ b/doc/pub/week35/html/._week35-bs031.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -353,7 +363,7 @@ $$
  • 40
  • 41
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs032.html b/doc/pub/week35/html/._week35-bs032.html index 4bf166711..3c6ff07dd 100644 --- a/doc/pub/week35/html/._week35-bs032.html +++ b/doc/pub/week35/html/._week35-bs032.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -364,7 +374,7 @@ $$
  • 41
  • 42
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs033.html b/doc/pub/week35/html/._week35-bs033.html index ca3a84a43..d0293d227 100644 --- a/doc/pub/week35/html/._week35-bs033.html +++ b/doc/pub/week35/html/._week35-bs033.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -388,7 +398,7 @@ reduced to the statistically relevant features.
  • 42
  • 43
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs034.html b/doc/pub/week35/html/._week35-bs034.html index ecd7997e9..28cadce7f 100644 --- a/doc/pub/week35/html/._week35-bs034.html +++ b/doc/pub/week35/html/._week35-bs034.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -393,7 +403,7 @@ This is equivalent to saying that the matrix \( \boldsymbol{X} \) has at least a
  • 43
  • 44
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs035.html b/doc/pub/week35/html/._week35-bs035.html index 202ab4f79..2bfd7304f 100644 --- a/doc/pub/week35/html/._week35-bs035.html +++ b/doc/pub/week35/html/._week35-bs035.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -368,7 +378,7 @@ $$
  • 44
  • 45
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs036.html b/doc/pub/week35/html/._week35-bs036.html index 218cc6d5c..861d9e943 100644 --- a/doc/pub/week35/html/._week35-bs036.html +++ b/doc/pub/week35/html/._week35-bs036.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -398,7 +408,7 @@ $$
  • 45
  • 46
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs037.html b/doc/pub/week35/html/._week35-bs037.html index 7f210a31f..e7908571d 100644 --- a/doc/pub/week35/html/._week35-bs037.html +++ b/doc/pub/week35/html/._week35-bs037.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -490,7 +500,7 @@ What happens if we do not include the intercept in our fit? We will discuss this
  • 46
  • 47
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs038.html b/doc/pub/week35/html/._week35-bs038.html index 9b11d89e2..fba642099 100644 --- a/doc/pub/week35/html/._week35-bs038.html +++ b/doc/pub/week35/html/._week35-bs038.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -380,7 +390,7 @@ $$
  • 47
  • 48
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs039.html b/doc/pub/week35/html/._week35-bs039.html index c7867b222..6296de721 100644 --- a/doc/pub/week35/html/._week35-bs039.html +++ b/doc/pub/week35/html/._week35-bs039.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -391,7 +401,7 @@ near singular or singular matrices.
  • 48
  • 49
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs040.html b/doc/pub/week35/html/._week35-bs040.html index 52b5406e8..f451c471c 100644 --- a/doc/pub/week35/html/._week35-bs040.html +++ b/doc/pub/week35/html/._week35-bs040.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -366,7 +376,7 @@ In general the economy-size SVD leads to less FLOPS and still conserving the des
  • 49
  • 50
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs041.html b/doc/pub/week35/html/._week35-bs041.html index a070820d7..05f7059ed 100644 --- a/doc/pub/week35/html/._week35-bs041.html +++ b/doc/pub/week35/html/._week35-bs041.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -407,7 +417,7 @@ in the program terminating due to a singular matrix.
  • 50
  • 51
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs042.html b/doc/pub/week35/html/._week35-bs042.html index 2cd6d1b23..aed38f6bc 100644 --- a/doc/pub/week35/html/._week35-bs042.html +++ b/doc/pub/week35/html/._week35-bs042.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -362,7 +372,7 @@ example
  • 51
  • 52
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs043.html b/doc/pub/week35/html/._week35-bs043.html index 751040cda..a31ddd276 100644 --- a/doc/pub/week35/html/._week35-bs043.html +++ b/doc/pub/week35/html/._week35-bs043.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -377,7 +387,7 @@ $$
  • 52
  • 53
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs044.html b/doc/pub/week35/html/._week35-bs044.html index 7a6c1cbad..655af893e 100644 --- a/doc/pub/week35/html/._week35-bs044.html +++ b/doc/pub/week35/html/._week35-bs044.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -401,7 +411,7 @@ decomposition of the design matrix.
  • 53
  • 54
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs045.html b/doc/pub/week35/html/._week35-bs045.html index 57965c4a3..52d0b5ba2 100644 --- a/doc/pub/week35/html/._week35-bs045.html +++ b/doc/pub/week35/html/._week35-bs045.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -388,7 +398,7 @@ orthogonality relation for the matrix \( \boldsymbol{U} \).
  • 54
  • 55
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs046.html b/doc/pub/week35/html/._week35-bs046.html index 83e41edf6..a4fd9037c 100644 --- a/doc/pub/week35/html/._week35-bs046.html +++ b/doc/pub/week35/html/._week35-bs046.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -392,7 +402,7 @@ for the definition of for example the covariance matrix and its relation to the
  • 55
  • 56
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs047.html b/doc/pub/week35/html/._week35-bs047.html index fa7360c2c..7f5872fee 100644 --- a/doc/pub/week35/html/._week35-bs047.html +++ b/doc/pub/week35/html/._week35-bs047.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -371,7 +381,7 @@ terms of the singular values. Let us develop these arguments, as they will pla
  • 56
  • 57
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs048.html b/doc/pub/week35/html/._week35-bs048.html index ae680dd81..6de8d0db4 100644 --- a/doc/pub/week35/html/._week35-bs048.html +++ b/doc/pub/week35/html/._week35-bs048.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -386,7 +396,7 @@ quantity will be computed with a factor \( 1/(n-1) \).
  • 57
  • 58
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs049.html b/doc/pub/week35/html/._week35-bs049.html index 18933937e..bc931d6a4 100644 --- a/doc/pub/week35/html/._week35-bs049.html +++ b/doc/pub/week35/html/._week35-bs049.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -371,7 +381,7 @@ $$
  • 58
  • 59
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs050.html b/doc/pub/week35/html/._week35-bs050.html index 641905ec1..dd95ad499 100644 --- a/doc/pub/week35/html/._week35-bs050.html +++ b/doc/pub/week35/html/._week35-bs050.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -404,7 +414,7 @@ $$
  • 59
  • 60
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs051.html b/doc/pub/week35/html/._week35-bs051.html index 9714b6ad2..c8681fb73 100644 --- a/doc/pub/week35/html/._week35-bs051.html +++ b/doc/pub/week35/html/._week35-bs051.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -400,7 +410,7 @@ C = np.c
  • 60
  • 61
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs052.html b/doc/pub/week35/html/._week35-bs052.html index 51a79daf9..c0bd95785 100644 --- a/doc/pub/week35/html/._week35-bs052.html +++ b/doc/pub/week35/html/._week35-bs052.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -402,6 +412,8 @@ this matrix we easily see that it is a positive definite matrix.
  • 60
  • 61
  • 62
  • +
  • ...
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs053.html b/doc/pub/week35/html/._week35-bs053.html index ccc0caa02..face2d88c 100644 --- a/doc/pub/week35/html/._week35-bs053.html +++ b/doc/pub/week35/html/._week35-bs053.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -382,6 +392,7 @@ correlation_matrix = Xpd60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs054.html b/doc/pub/week35/html/._week35-bs054.html index 8e4a66991..44f29548a 100644 --- a/doc/pub/week35/html/._week35-bs054.html +++ b/doc/pub/week35/html/._week35-bs054.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -377,6 +387,7 @@ $$
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs055.html b/doc/pub/week35/html/._week35-bs055.html index b8a0ed473..061f8655d 100644 --- a/doc/pub/week35/html/._week35-bs055.html +++ b/doc/pub/week35/html/._week35-bs055.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -374,6 +384,7 @@ $$
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs056.html b/doc/pub/week35/html/._week35-bs056.html index fff321b7b..02a8451fb 100644 --- a/doc/pub/week35/html/._week35-bs056.html +++ b/doc/pub/week35/html/._week35-bs056.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -379,6 +389,7 @@ absolute value of the eigenvalues of \( \boldsymbol{X} \).
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs057.html b/doc/pub/week35/html/._week35-bs057.html index 4895439fa..449912fe8 100644 --- a/doc/pub/week35/html/._week35-bs057.html +++ b/doc/pub/week35/html/._week35-bs057.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -372,6 +382,7 @@ values and the column vectors of \( \boldsymbol{V} \).
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs058.html b/doc/pub/week35/html/._week35-bs058.html index 51ddfe207..9f1761ec6 100644 --- a/doc/pub/week35/html/._week35-bs058.html +++ b/doc/pub/week35/html/._week35-bs058.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -412,6 +422,7 @@ $$
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs059.html b/doc/pub/week35/html/._week35-bs059.html index c9826857c..33475c612 100644 --- a/doc/pub/week35/html/._week35-bs059.html +++ b/doc/pub/week35/html/._week35-bs059.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -353,6 +363,7 @@ eigenvalues ordered in a descending way, that is \( \sigma_i \geq
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs060.html b/doc/pub/week35/html/._week35-bs060.html index 1786f345c..6804f90f6 100644 --- a/doc/pub/week35/html/._week35-bs060.html +++ b/doc/pub/week35/html/._week35-bs060.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -365,6 +375,7 @@ Similarly, Mehta et al
  • 60
  • 61
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs061.html b/doc/pub/week35/html/._week35-bs061.html index 8b084197b..2e0337a15 100644 --- a/doc/pub/week35/html/._week35-bs061.html +++ b/doc/pub/week35/html/._week35-bs061.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({
  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -365,6 +375,8 @@ $$
  • 60
  • 61
  • 62
  • +
  • 63
  • +
  • »
  • diff --git a/doc/pub/week35/html/._week35-bs062.html b/doc/pub/week35/html/._week35-bs062.html index a5c6a9056..0622367f7 100644 --- a/doc/pub/week35/html/._week35-bs062.html +++ b/doc/pub/week35/html/._week35-bs062.html @@ -65,7 +65,6 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'the-mean-squared-error-and-its-derivative'), - ('Other useful relations', 2, None, 'other-useful-relations'), ('Meet the Hessian Matrix', 2, None, 'meet-the-hessian-matrix'), ('Interpretations and optimizing our parameters', 2, @@ -212,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -252,8 +259,8 @@ MathJax.Hub.Config({
  • Reminder from last week
  • The equations for ordinary least squares
  • The cost/loss function
  • -
  • Interpretations and optimizing our parameters
  • -
  • Interpretations and optimizing our parameters
  • +
  • Interpretations and optimizing our parameters
  • +
  • Interpretations and optimizing our parameters
  • Some useful matrix and vector expressions
  • The Jacobian
  • Derivatives, example 1
  • @@ -261,55 +268,56 @@ MathJax.Hub.Config({
  • Example 3
  • Example 4
  • The mean squared error and its derivative
  • -
  • Other useful relations
  • -
  • Meet the Hessian Matrix
  • -
  • Interpretations and optimizing our parameters
  • -
  • Example relevant for the exercises
  • -
  • Own code for Ordinary Least Squares
  • -
  • Adding error analysis and training set up
  • -
  • Splitting our Data in Training and Test data
  • -
  • The complete code with a simple data set
  • -
  • Making your own test-train splitting
  • -
  • Reducing the number of degrees of freedom, overarching view
  • -
  • Preprocessing our data
  • -
  • Functionality in Scikit-Learn
  • -
  • More preprocessing
  • -
  • Frequently used scaling functions
  • -
  • Example of own Standard scaling
  • -
  • Min-Max Scaling
  • -
  • Testing the Means Squared Error as function of Complexity
  • -
  • Mathematical Interpretation of Ordinary Least Squares
  • -
  • Residual Error
  • -
  • Simple case
  • -
  • The singular value decomposition
  • -
  • Linear Regression Problems
  • -
  • Fixing the singularity
  • -
  • Ridge and LASSO Regression
  • -
  • Deriving the Ridge Regression Equations
  • -
  • Basic math of the SVD
  • -
  • The SVD, a Fantastic Algorithm
  • -
  • Economy-size SVD
  • -
  • Codes for the SVD
  • -
  • Note about SVD Calculations
  • -
  • Mathematics of the SVD and implications
  • -
  • Example Matrix
  • -
  • Setting up the Matrix to be inverted
  • -
  • Further properties (important for our analyses later)
  • -
  • Meet the Covariance Matrix
  • -
  • Introducing the Covariance and Correlation functions
  • -
  • Covariance and Correlation Matrix
  • -
  • Correlation Function and Design/Feature Matrix
  • -
  • Covariance Matrix Examples
  • -
  • Correlation Matrix
  • -
  • Correlation Matrix with Pandas
  • -
  • Rewriting the Covariance and/or Correlation Matrix
  • -
  • Linking with the SVD
  • -
  • What does it mean?
  • -
  • And finally \( \boldsymbol{X}\boldsymbol{X}^T \)
  • -
  • Back to Ridge and LASSO Regression
  • -
  • Interpreting the Ridge results
  • -
  • More interpretations
  • -
  • Deriving the Lasso Regression Equations
  • +
  • Meet the Hessian Matrix
  • +
  • Interpretations and optimizing our parameters
  • +
  • Example relevant for the exercises
  • +
  • Own code for Ordinary Least Squares
  • +
  • Adding error analysis and training set up
  • +
  • Splitting our Data in Training and Test data
  • +
  • The complete code with a simple data set
  • +
  • Making your own test-train splitting
  • +
  • Reducing the number of degrees of freedom, overarching view
  • +
  • Preprocessing our data
  • +
  • Functionality in Scikit-Learn
  • +
  • More preprocessing
  • +
  • Frequently used scaling functions
  • +
  • Example of own Standard scaling
  • +
  • Min-Max Scaling
  • +
  • Testing the Means Squared Error as function of Complexity
  • +
  • Mathematical Interpretation of Ordinary Least Squares
  • +
  • Residual Error
  • +
  • Simple case
  • +
  • The singular value decomposition
  • +
  • Linear Regression Problems
  • +
  • Fixing the singularity
  • +
  • Ridge and LASSO Regression
  • +
  • Deriving the Ridge Regression Equations
  • +
  • Basic math of the SVD
  • +
  • The SVD, a Fantastic Algorithm
  • +
  • Economy-size SVD
  • +
  • Codes for the SVD
  • +
  • Note about SVD Calculations
  • +
  • Mathematics of the SVD and implications
  • +
  • Example Matrix
  • +
  • Setting up the Matrix to be inverted
  • +
  • Further properties (important for our analyses later)
  • +
  • Meet the Covariance Matrix
  • +
  • Introducing the Covariance and Correlation functions
  • +
  • Covariance and Correlation Matrix
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Covariance Matrix Examples
  • +
  • Correlation Matrix
  • +
  • Correlation Matrix with Pandas
  • +
  • Rewriting the Covariance and/or Correlation Matrix
  • +
  • Linking with the SVD
  • +
  • What does it mean?
  • +
  • And finally \( \boldsymbol{X}\boldsymbol{X}^T \)
  • +
  • Back to Ridge and LASSO Regression
  • +
  • Interpreting the Ridge results
  • +
  • More interpretations
  • +
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -321,36 +329,555 @@ MathJax.Hub.Config({

     

     

     

    -

    Deriving the Lasso Regression Equations

    +

    Material for exercises week 35

    +

    Important technicalities: More on Rescaling data

    -

    Using the matrix-vector expression for Lasso regression, we have the following cost function

    +

    When you are comparing your own code with for example Scikit-Learn's +library, there are some technicalities to keep in mind. The examples +here demonstrate some of these aspects with potential pitfalls. +

    +

    The discussion here focuses on the role of the intercept, how we can +set up the design matrix, what scaling we should use and other topics +which tend confuse us. +

    + +

    The intercept can be interpreted as the expected value of our +target/output variables when all other predictors are set to zero. +Thus, if we cannot assume that the expected outputs/targets are zero +when all predictors are zero (the columns in the design matrix), it +may be a bad idea to implement a model which penalizes the intercept. +Furthermore, in for example Ridge and Lasso regression, the default solutions +from the library Scikit-Learn (when not shrinking \( \beta_0 \)) for the unknown parameters +\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and +\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values. +

    + +

    If our predictors represent different scales, then it is important to +standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each +column from the corresponding column and dividing the column with its +standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library, +the results may differ. +

    + +

    The +Standardscaler +function in Scikit-Learn does this for us. For the data sets we +have been studying in our various examples, the data are in many cases +already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a +survey of your data, with a critical assessment of them in case you need to scale the data. +

    + +

    If you need to scale the data, not doing so will give an unfair +penalization of the parameters since their magnitude depends on the +scale of their corresponding predictor. +

    + +

    The Scikit-Learn site https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section has a good discussion of different ways of preprocessing data.

    + +

    Suppose as an example that you +you have an input variable given by the heights of different persons. +Human height might be measured in inches or meters or +kilometers. If measured in kilometers, a standard linear regression +model with this predictor would probably give a much bigger +coefficient term, than if measured in millimeters. +This can clearly lead to problems in evaluating the cost/loss functions. +

    + +

    Keep in mind that when you transform your data set before training a model, the same transformation needs to be done +on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as +

    + + + +
    +
    +
    +
    +
    +
    """
    +#Model training, we compute the mean value of y and X
    +y_train_mean = np.mean(y_train)
    +X_train_mean = np.mean(X_train,axis=0)
    +X_train = X_train - X_train_mean
    +y_train = y_train - y_train_mean
    +
    +# The we fit our model with the training data
    +trained_model = some_model.fit(X_train,y_train)
    +
    +
    +#Model prediction, we need also to transform our data set used for the prediction.
    +X_test = X_test - X_train_mean #Use mean from training data
    +y_pred = trained_model(X_test)
    +y_pred = y_pred + y_train_mean
    +"""
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    Let us try to understand what this may imply mathematically when we +subtract the mean values, also known as zero centering. For +simplicity, we will focus on ordinary regression, as done in the above example. +

    + +

    The cost/loss function for regression is

    $$ -C(\boldsymbol{X},\boldsymbol{\theta})=\frac{1}{n}\left\{(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})^T(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})\right\}+\lambda\vert\vert\boldsymbol{\theta}\vert\vert_1, +C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,. $$ -

    Taking the derivative with respect to \( \boldsymbol{\theta} \) and recalling that the derivative of the absolute value is (we drop the boldfaced vector symbol for simplicty)

    -$$ -\frac{d \vert \theta\vert}{d \theta}=\mathrm{sgn}(\theta)=\left\{\begin{array}{cc} 1 & \theta > 0 \\-1 & \theta < 0, \end{array}\right. -$$ +

    Recall also that we use the squared value. This expression can lead to an +increased penalty for higher differences between predicted and +output/target values. +

    -

    we have that the derivative of the cost function is

    +

    What we have done is to single out the \( \beta_0 \) term in the +definition of the mean squared error (MSE). The design matrix \( X \) +does in this case not contain any intercept column. When we take the +derivative with respect to \( \beta_0 \), we want the derivative to obey +

    $$ -\frac{\partial C(\boldsymbol{X},\boldsymbol{\theta})}{\partial \boldsymbol{\theta}}=-\frac{2}{n}\boldsymbol{X}^T(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})+\lambda sgn(\boldsymbol{\theta})=0, +\frac{\partial C}{\partial \beta_j} = 0, $$ -

    and reordering we have

    +

    for all \( j \). For \( \beta_0 \) we have

    + $$ -\boldsymbol{X}^T\boldsymbol{X}\boldsymbol{\theta}+\frac{n}{2}\lambda sgn(\boldsymbol{\theta})=2\boldsymbol{X}^T\boldsymbol{y}. +\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right). $$ -

    We can redefine \( \lambda \) to absorb the constant \( n/2 \) and we rewrite the last equation as

    +

    Multiplying away the constant \( 2/n \), we obtain

    $$ -\boldsymbol{X}^T\boldsymbol{X}\boldsymbol{\theta}+\lambda sgn(\boldsymbol{\theta})=2\boldsymbol{X}^T\boldsymbol{y}. +\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j. $$ -

    This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package CVXOPT. We will discuss how to code LASSO regression next week, when we have introduced gradient methods.

    +

    Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \). +Our result for \( \beta_0 \) simplifies then to +

    +$$ +n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1. +$$ + +

    We obtain then

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}. +$$ + +

    If we define

    +$$ +\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}, +$$ + +

    and the mean value of the outputs as

    +$$ +\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i, +$$ + +

    we have

    +$$ +\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}. +$$ + +

    In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j. +$$ + +

    We can rewrite the latter equation as

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j, +$$ + +

    where we have defined

    +$$ +\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij}, +$$ + +

    the mean value for all elements of the column vector \( \boldsymbol{x}_j \).

    + +

    Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)

    +$$ +C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}). +$$ + +

    If we minimize with respect to \( \boldsymbol{\beta} \) we have then

    + +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}, +$$ + +

    where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \) +and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \). +

    + +

    For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then

    +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}. +$$ + +

    What does this mean? And why do we insist on all this? Let us look at some examples.

    + +

    This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (code example thanks to Øyvind Sigmundson Schøyen). Here our scaling of the data is done by subtracting the mean values only. +Note also that we do not split the data into training and test. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import matplotlib.pyplot as plt
    +
    +from sklearn.linear_model import LinearRegression
    +
    +
    +np.random.seed(2021)
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +def fit_beta(X, y):
    +    return np.linalg.pinv(X.T @ X) @ X.T @ y
    +
    +
    +true_beta = [2, 0.5, 3.7]
    +
    +x = np.linspace(0, 1, 11)
    +y = np.sum(
    +    np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0
    +) + 0.1 * np.random.normal(size=len(x))
    +
    +degree = 3
    +X = np.zeros((len(x), degree))
    +
    +# Include the intercept in the design matrix
    +for p in range(degree):
    +    X[:, p] = x ** p
    +
    +beta = fit_beta(X, y)
    +
    +# Intercept is included in the design matrix
    +skl = LinearRegression(fit_intercept=False).fit(X, y)
    +
    +print(f"True beta: {true_beta}")
    +print(f"Fitted beta: {beta}")
    +print(f"Sklearn fitted beta: {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with intercept column")
    +print(MSE(y,ypredictOwn))
    +print(f"MSE with intercept column from SKL")
    +print(MSE(y,ypredictSKL))
    +
    +
    +plt.figure()
    +plt.scatter(x, y, label="Data")
    +plt.plot(x, X @ beta, label="Fit")
    +plt.plot(x, skl.predict(X), label="Sklearn (fit_intercept=False)")
    +
    +
    +# Do not include the intercept in the design matrix
    +X = np.zeros((len(x), degree - 1))
    +
    +for p in range(degree - 1):
    +    X[:, p] = x ** (p + 1)
    +
    +# Intercept is not included in the design matrix
    +skl = LinearRegression(fit_intercept=True).fit(X, y)
    +
    +# Use centered values for X and y when computing coefficients
    +y_offset = np.average(y, axis=0)
    +X_offset = np.average(X, axis=0)
    +
    +beta = fit_beta(X - X_offset, y - y_offset)
    +intercept = np.mean(y_offset - X_offset @ beta)
    +
    +print(f"Manual intercept: {intercept}")
    +print(f"Fitted beta (without intercept): {beta}")
    +print(f"Sklearn intercept: {skl.intercept_}")
    +print(f"Sklearn fitted beta (without intercept): {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with Manual intercept")
    +print(MSE(y,ypredictOwn+intercept))
    +print(f"MSE with Sklearn intercept")
    +print(MSE(y,ypredictSKL))
    +
    +plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
    +plt.plot(x, skl.predict(X), "--", label="Sklearn (fit_intercept=True)")
    +plt.grid()
    +plt.legend()
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). +

    + +

    Printing the MSE, we see first that both methods give the same MSE, as +they should. However, when we move to for example Ridge regression, +the way we treat the intercept may give a larger or smaller MSE, +meaning that the MSE can be penalized by the value of the +intercept. Not including the intercept in the fit, means that the +regularization term does not include \( \beta_0 \). For different values +of \( \lambda \), this may lead to different MSE values. +

    + +

    To remind the reader, the regularization term, with the intercept in Ridge regression, is given by

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2, +$$ + +

    but when we take out the intercept, this equation becomes

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2. +$$ + +

    For Lasso regression we have

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert. +$$ + +

    It means that, when scaling the design matrix and the outputs/targets, +by subtracting the mean values, we have an optimization problem which +is not penalized by the intercept. The MSE value can then be smaller +since it focuses only on the remaining quantities. If we however bring +back the intercept, we will get a MSE which then contains the +intercept. +

    + +

    Armed with this wisdom, we attempt first to simply set the intercept equal to False in our implementation of Ridge regression for our well-known vanilla data set.

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(3155)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree))
    +#We include explicitely the intercept column
    +for degree in range(Maxpolydegree):
    +    X[:,degree] = x**degree
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +p = Maxpolydegree
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
    +    # Note: we include the intercept column and no scaling
    +    RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
    +    RegRidge.fit(X_train,y_train)
    +    # and then make the prediction
    +    ytildeOwnRidge = X_train @ OwnRidgeBeta
    +    ypredictOwnRidge = X_test @ OwnRidgeBeta
    +    ytildeRidge = RegRidge.predict(X_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta)
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')
    +
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. +We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix. +What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +from sklearn.preprocessing import StandardScaler
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(315)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree-1))
    +
    +for degree in range(1,Maxpolydegree): #No intercept column
    +    X[:,degree-1] = x**(degree)
    +
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
    +X_train_mean = np.mean(X_train,axis=0)
    +#Center by removing mean from each feature
    +X_train_scaled = X_train - X_train_mean 
    +X_test_scaled = X_test - X_train_mean
    +#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
    +#Remove the intercept from the training data.
    +y_scaler = np.mean(y_train)           
    +y_train_scaled = y_train - y_scaler   
    +
    +p = Maxpolydegree-1
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
    +    intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
    +    #Add intercept to prediction
    +    ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler 
    +    RegRidge = linear_model.Ridge(lmb)
    +    RegRidge.fit(X_train,y_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta) #Intercept is given by mean of target variable
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print('Intercept from own implementation:')
    +    print(intercept_)
    +    print('Intercept from Scikit-Learn Ridge implementation')
    +    print(RegRidge.intercept_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    We see here, when compared to the code which includes explicitely the +intercept column, that our MSE value is actually smaller. This is +because the regularization term does not include the intercept value +\( \beta_0 \) in the fitting. This applies to Lasso regularization as +well. It means that our optimization is now done only with the +centered matrix and/or vector that enter the fitting procedure. +

    diff --git a/doc/pub/week35/html/week35-bs.html b/doc/pub/week35/html/week35-bs.html index 028edad4c..ad2ecd4da 100644 --- a/doc/pub/week35/html/week35-bs.html +++ b/doc/pub/week35/html/week35-bs.html @@ -211,7 +211,15 @@ doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=d ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -308,6 +316,8 @@ MathJax.Hub.Config({

  • Interpreting the Ridge results
  • More interpretations
  • Deriving the Lasso Regression Equations
  • +
  • Material for exercises week 35
  • +
  • Important technicalities: More on Rescaling data
  • @@ -359,7 +369,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 62
  • +
  • 63
  • »
  • diff --git a/doc/pub/week35/html/week35-reveal.html b/doc/pub/week35/html/week35-reveal.html index 6548291c3..fdeb4ebb3 100644 --- a/doc/pub/week35/html/week35-reveal.html +++ b/doc/pub/week35/html/week35-reveal.html @@ -2996,6 +2996,594 @@ $$

    This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package CVXOPT. We will discuss how to code LASSO regression next week, when we have introduced gradient methods.

    +
    +

    Material for exercises week 35

    +

    Important technicalities: More on Rescaling data

    + +

    When you are comparing your own code with for example Scikit-Learn's +library, there are some technicalities to keep in mind. The examples +here demonstrate some of these aspects with potential pitfalls. +

    + +

    The discussion here focuses on the role of the intercept, how we can +set up the design matrix, what scaling we should use and other topics +which tend confuse us. +

    + +

    The intercept can be interpreted as the expected value of our +target/output variables when all other predictors are set to zero. +Thus, if we cannot assume that the expected outputs/targets are zero +when all predictors are zero (the columns in the design matrix), it +may be a bad idea to implement a model which penalizes the intercept. +Furthermore, in for example Ridge and Lasso regression, the default solutions +from the library Scikit-Learn (when not shrinking \( \beta_0 \)) for the unknown parameters +\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and +\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values. +

    + +

    If our predictors represent different scales, then it is important to +standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each +column from the corresponding column and dividing the column with its +standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library, +the results may differ. +

    + +

    The +Standardscaler +function in Scikit-Learn does this for us. For the data sets we +have been studying in our various examples, the data are in many cases +already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a +survey of your data, with a critical assessment of them in case you need to scale the data. +

    + +

    If you need to scale the data, not doing so will give an unfair +penalization of the parameters since their magnitude depends on the +scale of their corresponding predictor. +

    + +

    The Scikit-Learn site https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section has a good discussion of different ways of preprocessing data.

    + +

    Suppose as an example that you +you have an input variable given by the heights of different persons. +Human height might be measured in inches or meters or +kilometers. If measured in kilometers, a standard linear regression +model with this predictor would probably give a much bigger +coefficient term, than if measured in millimeters. +This can clearly lead to problems in evaluating the cost/loss functions. +

    + +

    Keep in mind that when you transform your data set before training a model, the same transformation needs to be done +on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as +

    + + + +
    +
    +
    +
    +
    +
    """
    +#Model training, we compute the mean value of y and X
    +y_train_mean = np.mean(y_train)
    +X_train_mean = np.mean(X_train,axis=0)
    +X_train = X_train - X_train_mean
    +y_train = y_train - y_train_mean
    +
    +# The we fit our model with the training data
    +trained_model = some_model.fit(X_train,y_train)
    +
    +
    +#Model prediction, we need also to transform our data set used for the prediction.
    +X_test = X_test - X_train_mean #Use mean from training data
    +y_pred = trained_model(X_test)
    +y_pred = y_pred + y_train_mean
    +"""
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    Let us try to understand what this may imply mathematically when we +subtract the mean values, also known as zero centering. For +simplicity, we will focus on ordinary regression, as done in the above example. +

    + +

    The cost/loss function for regression is

    +

     
    +$$ +C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,. +$$ +

     
    + +

    Recall also that we use the squared value. This expression can lead to an +increased penalty for higher differences between predicted and +output/target values. +

    + +

    What we have done is to single out the \( \beta_0 \) term in the +definition of the mean squared error (MSE). The design matrix \( X \) +does in this case not contain any intercept column. When we take the +derivative with respect to \( \beta_0 \), we want the derivative to obey +

    + +

     
    +$$ +\frac{\partial C}{\partial \beta_j} = 0, +$$ +

     
    + +

    for all \( j \). For \( \beta_0 \) we have

    + +

     
    +$$ +\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right). +$$ +

     
    + +

    Multiplying away the constant \( 2/n \), we obtain

    +

     
    +$$ +\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j. +$$ +

     
    + +

    Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \). +Our result for \( \beta_0 \) simplifies then to +

    +

     
    +$$ +n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1. +$$ +

     
    + +

    We obtain then

    +

     
    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}. +$$ +

     
    + +

    If we define

    +

     
    +$$ +\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}, +$$ +

     
    + +

    and the mean value of the outputs as

    +

     
    +$$ +\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i, +$$ +

     
    + +

    we have

    +

     
    +$$ +\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}. +$$ +

     
    + +

    In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have

    +

     
    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j. +$$ +

     
    + +

    We can rewrite the latter equation as

    +

     
    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j, +$$ +

     
    + +

    where we have defined

    +

     
    +$$ +\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij}, +$$ +

     
    + +

    the mean value for all elements of the column vector \( \boldsymbol{x}_j \).

    + +

    Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)

    +

     
    +$$ +C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}). +$$ +

     
    + +

    If we minimize with respect to \( \boldsymbol{\beta} \) we have then

    + +

     
    +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}, +$$ +

     
    + +

    where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \) +and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \). +

    + +

    For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then

    +

     
    +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}. +$$ +

     
    + +

    What does this mean? And why do we insist on all this? Let us look at some examples.

    + +

    This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (code example thanks to Øyvind Sigmundson Schøyen). Here our scaling of the data is done by subtracting the mean values only. +Note also that we do not split the data into training and test. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import matplotlib.pyplot as plt
    +
    +from sklearn.linear_model import LinearRegression
    +
    +
    +np.random.seed(2021)
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +def fit_beta(X, y):
    +    return np.linalg.pinv(X.T @ X) @ X.T @ y
    +
    +
    +true_beta = [2, 0.5, 3.7]
    +
    +x = np.linspace(0, 1, 11)
    +y = np.sum(
    +    np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0
    +) + 0.1 * np.random.normal(size=len(x))
    +
    +degree = 3
    +X = np.zeros((len(x), degree))
    +
    +# Include the intercept in the design matrix
    +for p in range(degree):
    +    X[:, p] = x ** p
    +
    +beta = fit_beta(X, y)
    +
    +# Intercept is included in the design matrix
    +skl = LinearRegression(fit_intercept=False).fit(X, y)
    +
    +print(f"True beta: {true_beta}")
    +print(f"Fitted beta: {beta}")
    +print(f"Sklearn fitted beta: {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with intercept column")
    +print(MSE(y,ypredictOwn))
    +print(f"MSE with intercept column from SKL")
    +print(MSE(y,ypredictSKL))
    +
    +
    +plt.figure()
    +plt.scatter(x, y, label="Data")
    +plt.plot(x, X @ beta, label="Fit")
    +plt.plot(x, skl.predict(X), label="Sklearn (fit_intercept=False)")
    +
    +
    +# Do not include the intercept in the design matrix
    +X = np.zeros((len(x), degree - 1))
    +
    +for p in range(degree - 1):
    +    X[:, p] = x ** (p + 1)
    +
    +# Intercept is not included in the design matrix
    +skl = LinearRegression(fit_intercept=True).fit(X, y)
    +
    +# Use centered values for X and y when computing coefficients
    +y_offset = np.average(y, axis=0)
    +X_offset = np.average(X, axis=0)
    +
    +beta = fit_beta(X - X_offset, y - y_offset)
    +intercept = np.mean(y_offset - X_offset @ beta)
    +
    +print(f"Manual intercept: {intercept}")
    +print(f"Fitted beta (without intercept): {beta}")
    +print(f"Sklearn intercept: {skl.intercept_}")
    +print(f"Sklearn fitted beta (without intercept): {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with Manual intercept")
    +print(MSE(y,ypredictOwn+intercept))
    +print(f"MSE with Sklearn intercept")
    +print(MSE(y,ypredictSKL))
    +
    +plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
    +plt.plot(x, skl.predict(X), "--", label="Sklearn (fit_intercept=True)")
    +plt.grid()
    +plt.legend()
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). +

    + +

    Printing the MSE, we see first that both methods give the same MSE, as +they should. However, when we move to for example Ridge regression, +the way we treat the intercept may give a larger or smaller MSE, +meaning that the MSE can be penalized by the value of the +intercept. Not including the intercept in the fit, means that the +regularization term does not include \( \beta_0 \). For different values +of \( \lambda \), this may lead to different MSE values. +

    + +

    To remind the reader, the regularization term, with the intercept in Ridge regression, is given by

    +

     
    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2, +$$ +

     
    + +

    but when we take out the intercept, this equation becomes

    +

     
    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2. +$$ +

     
    + +

    For Lasso regression we have

    +

     
    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert. +$$ +

     
    + +

    It means that, when scaling the design matrix and the outputs/targets, +by subtracting the mean values, we have an optimization problem which +is not penalized by the intercept. The MSE value can then be smaller +since it focuses only on the remaining quantities. If we however bring +back the intercept, we will get a MSE which then contains the +intercept. +

    + +

    Armed with this wisdom, we attempt first to simply set the intercept equal to False in our implementation of Ridge regression for our well-known vanilla data set.

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(3155)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree))
    +#We include explicitely the intercept column
    +for degree in range(Maxpolydegree):
    +    X[:,degree] = x**degree
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +p = Maxpolydegree
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
    +    # Note: we include the intercept column and no scaling
    +    RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
    +    RegRidge.fit(X_train,y_train)
    +    # and then make the prediction
    +    ytildeOwnRidge = X_train @ OwnRidgeBeta
    +    ypredictOwnRidge = X_test @ OwnRidgeBeta
    +    ytildeRidge = RegRidge.predict(X_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta)
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')
    +
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. +We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix. +What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +from sklearn.preprocessing import StandardScaler
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(315)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree-1))
    +
    +for degree in range(1,Maxpolydegree): #No intercept column
    +    X[:,degree-1] = x**(degree)
    +
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
    +X_train_mean = np.mean(X_train,axis=0)
    +#Center by removing mean from each feature
    +X_train_scaled = X_train - X_train_mean 
    +X_test_scaled = X_test - X_train_mean
    +#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
    +#Remove the intercept from the training data.
    +y_scaler = np.mean(y_train)           
    +y_train_scaled = y_train - y_scaler   
    +
    +p = Maxpolydegree-1
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
    +    intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
    +    #Add intercept to prediction
    +    ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler 
    +    RegRidge = linear_model.Ridge(lmb)
    +    RegRidge.fit(X_train,y_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta) #Intercept is given by mean of target variable
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print('Intercept from own implementation:')
    +    print(intercept_)
    +    print('Intercept from Scikit-Learn Ridge implementation')
    +    print(RegRidge.intercept_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    We see here, when compared to the code which includes explicitely the +intercept column, that our MSE value is actually smaller. This is +because the regularization term does not include the intercept value +\( \beta_0 \) in the fitting. This applies to Lasso regularization as +well. It means that our optimization is now done only with the +centered matrix and/or vector that enter the fitting procedure. +

    +
    + diff --git a/doc/pub/week35/html/week35-solarized.html b/doc/pub/week35/html/week35-solarized.html index 0501b9424..9dfe9620b 100644 --- a/doc/pub/week35/html/week35-solarized.html +++ b/doc/pub/week35/html/week35-solarized.html @@ -238,7 +238,15 @@ div.toc p,a { ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -2747,6 +2755,557 @@ $$

    This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package CVXOPT. We will discuss how to code LASSO regression next week, when we have introduced gradient methods.

    +









    +

    Material for exercises week 35

    +

    Important technicalities: More on Rescaling data

    + +

    When you are comparing your own code with for example Scikit-Learn's +library, there are some technicalities to keep in mind. The examples +here demonstrate some of these aspects with potential pitfalls. +

    + +

    The discussion here focuses on the role of the intercept, how we can +set up the design matrix, what scaling we should use and other topics +which tend confuse us. +

    + +

    The intercept can be interpreted as the expected value of our +target/output variables when all other predictors are set to zero. +Thus, if we cannot assume that the expected outputs/targets are zero +when all predictors are zero (the columns in the design matrix), it +may be a bad idea to implement a model which penalizes the intercept. +Furthermore, in for example Ridge and Lasso regression, the default solutions +from the library Scikit-Learn (when not shrinking \( \beta_0 \)) for the unknown parameters +\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and +\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values. +

    + +

    If our predictors represent different scales, then it is important to +standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each +column from the corresponding column and dividing the column with its +standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library, +the results may differ. +

    + +

    The +Standardscaler +function in Scikit-Learn does this for us. For the data sets we +have been studying in our various examples, the data are in many cases +already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a +survey of your data, with a critical assessment of them in case you need to scale the data. +

    + +

    If you need to scale the data, not doing so will give an unfair +penalization of the parameters since their magnitude depends on the +scale of their corresponding predictor. +

    + +

    The Scikit-Learn site https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section has a good discussion of different ways of preprocessing data.

    + +

    Suppose as an example that you +you have an input variable given by the heights of different persons. +Human height might be measured in inches or meters or +kilometers. If measured in kilometers, a standard linear regression +model with this predictor would probably give a much bigger +coefficient term, than if measured in millimeters. +This can clearly lead to problems in evaluating the cost/loss functions. +

    + +

    Keep in mind that when you transform your data set before training a model, the same transformation needs to be done +on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as +

    + + + +
    +
    +
    +
    +
    +
    """
    +#Model training, we compute the mean value of y and X
    +y_train_mean = np.mean(y_train)
    +X_train_mean = np.mean(X_train,axis=0)
    +X_train = X_train - X_train_mean
    +y_train = y_train - y_train_mean
    +
    +# The we fit our model with the training data
    +trained_model = some_model.fit(X_train,y_train)
    +
    +
    +#Model prediction, we need also to transform our data set used for the prediction.
    +X_test = X_test - X_train_mean #Use mean from training data
    +y_pred = trained_model(X_test)
    +y_pred = y_pred + y_train_mean
    +"""
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    Let us try to understand what this may imply mathematically when we +subtract the mean values, also known as zero centering. For +simplicity, we will focus on ordinary regression, as done in the above example. +

    + +

    The cost/loss function for regression is

    +$$ +C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,. +$$ + +

    Recall also that we use the squared value. This expression can lead to an +increased penalty for higher differences between predicted and +output/target values. +

    + +

    What we have done is to single out the \( \beta_0 \) term in the +definition of the mean squared error (MSE). The design matrix \( X \) +does in this case not contain any intercept column. When we take the +derivative with respect to \( \beta_0 \), we want the derivative to obey +

    + +$$ +\frac{\partial C}{\partial \beta_j} = 0, +$$ + +

    for all \( j \). For \( \beta_0 \) we have

    + +$$ +\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right). +$$ + +

    Multiplying away the constant \( 2/n \), we obtain

    +$$ +\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j. +$$ + +

    Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \). +Our result for \( \beta_0 \) simplifies then to +

    +$$ +n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1. +$$ + +

    We obtain then

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}. +$$ + +

    If we define

    +$$ +\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}, +$$ + +

    and the mean value of the outputs as

    +$$ +\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i, +$$ + +

    we have

    +$$ +\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}. +$$ + +

    In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j. +$$ + +

    We can rewrite the latter equation as

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j, +$$ + +

    where we have defined

    +$$ +\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij}, +$$ + +

    the mean value for all elements of the column vector \( \boldsymbol{x}_j \).

    + +

    Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)

    +$$ +C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}). +$$ + +

    If we minimize with respect to \( \boldsymbol{\beta} \) we have then

    + +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}, +$$ + +

    where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \) +and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \). +

    + +

    For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then

    +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}. +$$ + +

    What does this mean? And why do we insist on all this? Let us look at some examples.

    + +

    This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (code example thanks to Øyvind Sigmundson Schøyen). Here our scaling of the data is done by subtracting the mean values only. +Note also that we do not split the data into training and test. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import matplotlib.pyplot as plt
    +
    +from sklearn.linear_model import LinearRegression
    +
    +
    +np.random.seed(2021)
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +def fit_beta(X, y):
    +    return np.linalg.pinv(X.T @ X) @ X.T @ y
    +
    +
    +true_beta = [2, 0.5, 3.7]
    +
    +x = np.linspace(0, 1, 11)
    +y = np.sum(
    +    np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0
    +) + 0.1 * np.random.normal(size=len(x))
    +
    +degree = 3
    +X = np.zeros((len(x), degree))
    +
    +# Include the intercept in the design matrix
    +for p in range(degree):
    +    X[:, p] = x ** p
    +
    +beta = fit_beta(X, y)
    +
    +# Intercept is included in the design matrix
    +skl = LinearRegression(fit_intercept=False).fit(X, y)
    +
    +print(f"True beta: {true_beta}")
    +print(f"Fitted beta: {beta}")
    +print(f"Sklearn fitted beta: {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with intercept column")
    +print(MSE(y,ypredictOwn))
    +print(f"MSE with intercept column from SKL")
    +print(MSE(y,ypredictSKL))
    +
    +
    +plt.figure()
    +plt.scatter(x, y, label="Data")
    +plt.plot(x, X @ beta, label="Fit")
    +plt.plot(x, skl.predict(X), label="Sklearn (fit_intercept=False)")
    +
    +
    +# Do not include the intercept in the design matrix
    +X = np.zeros((len(x), degree - 1))
    +
    +for p in range(degree - 1):
    +    X[:, p] = x ** (p + 1)
    +
    +# Intercept is not included in the design matrix
    +skl = LinearRegression(fit_intercept=True).fit(X, y)
    +
    +# Use centered values for X and y when computing coefficients
    +y_offset = np.average(y, axis=0)
    +X_offset = np.average(X, axis=0)
    +
    +beta = fit_beta(X - X_offset, y - y_offset)
    +intercept = np.mean(y_offset - X_offset @ beta)
    +
    +print(f"Manual intercept: {intercept}")
    +print(f"Fitted beta (without intercept): {beta}")
    +print(f"Sklearn intercept: {skl.intercept_}")
    +print(f"Sklearn fitted beta (without intercept): {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with Manual intercept")
    +print(MSE(y,ypredictOwn+intercept))
    +print(f"MSE with Sklearn intercept")
    +print(MSE(y,ypredictSKL))
    +
    +plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
    +plt.plot(x, skl.predict(X), "--", label="Sklearn (fit_intercept=True)")
    +plt.grid()
    +plt.legend()
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). +

    + +

    Printing the MSE, we see first that both methods give the same MSE, as +they should. However, when we move to for example Ridge regression, +the way we treat the intercept may give a larger or smaller MSE, +meaning that the MSE can be penalized by the value of the +intercept. Not including the intercept in the fit, means that the +regularization term does not include \( \beta_0 \). For different values +of \( \lambda \), this may lead to different MSE values. +

    + +

    To remind the reader, the regularization term, with the intercept in Ridge regression, is given by

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2, +$$ + +

    but when we take out the intercept, this equation becomes

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2. +$$ + +

    For Lasso regression we have

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert. +$$ + +

    It means that, when scaling the design matrix and the outputs/targets, +by subtracting the mean values, we have an optimization problem which +is not penalized by the intercept. The MSE value can then be smaller +since it focuses only on the remaining quantities. If we however bring +back the intercept, we will get a MSE which then contains the +intercept. +

    + +

    Armed with this wisdom, we attempt first to simply set the intercept equal to False in our implementation of Ridge regression for our well-known vanilla data set.

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(3155)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree))
    +#We include explicitely the intercept column
    +for degree in range(Maxpolydegree):
    +    X[:,degree] = x**degree
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +p = Maxpolydegree
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
    +    # Note: we include the intercept column and no scaling
    +    RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
    +    RegRidge.fit(X_train,y_train)
    +    # and then make the prediction
    +    ytildeOwnRidge = X_train @ OwnRidgeBeta
    +    ypredictOwnRidge = X_test @ OwnRidgeBeta
    +    ytildeRidge = RegRidge.predict(X_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta)
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')
    +
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. +We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix. +What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +from sklearn.preprocessing import StandardScaler
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(315)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree-1))
    +
    +for degree in range(1,Maxpolydegree): #No intercept column
    +    X[:,degree-1] = x**(degree)
    +
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
    +X_train_mean = np.mean(X_train,axis=0)
    +#Center by removing mean from each feature
    +X_train_scaled = X_train - X_train_mean 
    +X_test_scaled = X_test - X_train_mean
    +#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
    +#Remove the intercept from the training data.
    +y_scaler = np.mean(y_train)           
    +y_train_scaled = y_train - y_scaler   
    +
    +p = Maxpolydegree-1
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
    +    intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
    +    #Add intercept to prediction
    +    ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler 
    +    RegRidge = linear_model.Ridge(lmb)
    +    RegRidge.fit(X_train,y_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta) #Intercept is given by mean of target variable
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print('Intercept from own implementation:')
    +    print(intercept_)
    +    print('Intercept from Scikit-Learn Ridge implementation')
    +    print(RegRidge.intercept_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    We see here, when compared to the code which includes explicitely the +intercept column, that our MSE value is actually smaller. This is +because the regularization term does not include the intercept value +\( \beta_0 \) in the fitting. This applies to Lasso regularization as +well. It means that our optimization is now done only with the +centered matrix and/or vector that enter the fitting procedure. +

    +
    © 1999-2025, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license diff --git a/doc/pub/week35/html/week35.html b/doc/pub/week35/html/week35.html index bc92c26d0..2d2eed06a 100644 --- a/doc/pub/week35/html/week35.html +++ b/doc/pub/week35/html/week35.html @@ -315,7 +315,15 @@ div.toc p,a { ('Deriving the Lasso Regression Equations', 2, None, - 'deriving-the-lasso-regression-equations')]} + 'deriving-the-lasso-regression-equations'), + ('Material for exercises week 35', + 2, + None, + 'material-for-exercises-week-35'), + ('Important technicalities: More on Rescaling data', + 2, + None, + 'important-technicalities-more-on-rescaling-data')]} end of tocinfo --> @@ -2824,6 +2832,557 @@ $$

    This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package CVXOPT. We will discuss how to code LASSO regression next week, when we have introduced gradient methods.

    +









    +

    Material for exercises week 35

    +

    Important technicalities: More on Rescaling data

    + +

    When you are comparing your own code with for example Scikit-Learn's +library, there are some technicalities to keep in mind. The examples +here demonstrate some of these aspects with potential pitfalls. +

    + +

    The discussion here focuses on the role of the intercept, how we can +set up the design matrix, what scaling we should use and other topics +which tend confuse us. +

    + +

    The intercept can be interpreted as the expected value of our +target/output variables when all other predictors are set to zero. +Thus, if we cannot assume that the expected outputs/targets are zero +when all predictors are zero (the columns in the design matrix), it +may be a bad idea to implement a model which penalizes the intercept. +Furthermore, in for example Ridge and Lasso regression, the default solutions +from the library Scikit-Learn (when not shrinking \( \beta_0 \)) for the unknown parameters +\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and +\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values. +

    + +

    If our predictors represent different scales, then it is important to +standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each +column from the corresponding column and dividing the column with its +standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library, +the results may differ. +

    + +

    The +Standardscaler +function in Scikit-Learn does this for us. For the data sets we +have been studying in our various examples, the data are in many cases +already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a +survey of your data, with a critical assessment of them in case you need to scale the data. +

    + +

    If you need to scale the data, not doing so will give an unfair +penalization of the parameters since their magnitude depends on the +scale of their corresponding predictor. +

    + +

    The Scikit-Learn site https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section has a good discussion of different ways of preprocessing data.

    + +

    Suppose as an example that you +you have an input variable given by the heights of different persons. +Human height might be measured in inches or meters or +kilometers. If measured in kilometers, a standard linear regression +model with this predictor would probably give a much bigger +coefficient term, than if measured in millimeters. +This can clearly lead to problems in evaluating the cost/loss functions. +

    + +

    Keep in mind that when you transform your data set before training a model, the same transformation needs to be done +on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as +

    + + + +
    +
    +
    +
    +
    +
    """
    +#Model training, we compute the mean value of y and X
    +y_train_mean = np.mean(y_train)
    +X_train_mean = np.mean(X_train,axis=0)
    +X_train = X_train - X_train_mean
    +y_train = y_train - y_train_mean
    +
    +# The we fit our model with the training data
    +trained_model = some_model.fit(X_train,y_train)
    +
    +
    +#Model prediction, we need also to transform our data set used for the prediction.
    +X_test = X_test - X_train_mean #Use mean from training data
    +y_pred = trained_model(X_test)
    +y_pred = y_pred + y_train_mean
    +"""
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    Let us try to understand what this may imply mathematically when we +subtract the mean values, also known as zero centering. For +simplicity, we will focus on ordinary regression, as done in the above example. +

    + +

    The cost/loss function for regression is

    +$$ +C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,. +$$ + +

    Recall also that we use the squared value. This expression can lead to an +increased penalty for higher differences between predicted and +output/target values. +

    + +

    What we have done is to single out the \( \beta_0 \) term in the +definition of the mean squared error (MSE). The design matrix \( X \) +does in this case not contain any intercept column. When we take the +derivative with respect to \( \beta_0 \), we want the derivative to obey +

    + +$$ +\frac{\partial C}{\partial \beta_j} = 0, +$$ + +

    for all \( j \). For \( \beta_0 \) we have

    + +$$ +\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right). +$$ + +

    Multiplying away the constant \( 2/n \), we obtain

    +$$ +\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j. +$$ + +

    Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \). +Our result for \( \beta_0 \) simplifies then to +

    +$$ +n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1. +$$ + +

    We obtain then

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}. +$$ + +

    If we define

    +$$ +\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}, +$$ + +

    and the mean value of the outputs as

    +$$ +\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i, +$$ + +

    we have

    +$$ +\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}. +$$ + +

    In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j. +$$ + +

    We can rewrite the latter equation as

    +$$ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j, +$$ + +

    where we have defined

    +$$ +\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij}, +$$ + +

    the mean value for all elements of the column vector \( \boldsymbol{x}_j \).

    + +

    Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)

    +$$ +C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}). +$$ + +

    If we minimize with respect to \( \boldsymbol{\beta} \) we have then

    + +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}, +$$ + +

    where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \) +and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \). +

    + +

    For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then

    +$$ +\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}. +$$ + +

    What does this mean? And why do we insist on all this? Let us look at some examples.

    + +

    This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (code example thanks to Øyvind Sigmundson Schøyen). Here our scaling of the data is done by subtracting the mean values only. +Note also that we do not split the data into training and test. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import matplotlib.pyplot as plt
    +
    +from sklearn.linear_model import LinearRegression
    +
    +
    +np.random.seed(2021)
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +def fit_beta(X, y):
    +    return np.linalg.pinv(X.T @ X) @ X.T @ y
    +
    +
    +true_beta = [2, 0.5, 3.7]
    +
    +x = np.linspace(0, 1, 11)
    +y = np.sum(
    +    np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0
    +) + 0.1 * np.random.normal(size=len(x))
    +
    +degree = 3
    +X = np.zeros((len(x), degree))
    +
    +# Include the intercept in the design matrix
    +for p in range(degree):
    +    X[:, p] = x ** p
    +
    +beta = fit_beta(X, y)
    +
    +# Intercept is included in the design matrix
    +skl = LinearRegression(fit_intercept=False).fit(X, y)
    +
    +print(f"True beta: {true_beta}")
    +print(f"Fitted beta: {beta}")
    +print(f"Sklearn fitted beta: {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with intercept column")
    +print(MSE(y,ypredictOwn))
    +print(f"MSE with intercept column from SKL")
    +print(MSE(y,ypredictSKL))
    +
    +
    +plt.figure()
    +plt.scatter(x, y, label="Data")
    +plt.plot(x, X @ beta, label="Fit")
    +plt.plot(x, skl.predict(X), label="Sklearn (fit_intercept=False)")
    +
    +
    +# Do not include the intercept in the design matrix
    +X = np.zeros((len(x), degree - 1))
    +
    +for p in range(degree - 1):
    +    X[:, p] = x ** (p + 1)
    +
    +# Intercept is not included in the design matrix
    +skl = LinearRegression(fit_intercept=True).fit(X, y)
    +
    +# Use centered values for X and y when computing coefficients
    +y_offset = np.average(y, axis=0)
    +X_offset = np.average(X, axis=0)
    +
    +beta = fit_beta(X - X_offset, y - y_offset)
    +intercept = np.mean(y_offset - X_offset @ beta)
    +
    +print(f"Manual intercept: {intercept}")
    +print(f"Fitted beta (without intercept): {beta}")
    +print(f"Sklearn intercept: {skl.intercept_}")
    +print(f"Sklearn fitted beta (without intercept): {skl.coef_}")
    +ypredictOwn = X @ beta
    +ypredictSKL = skl.predict(X)
    +print(f"MSE with Manual intercept")
    +print(MSE(y,ypredictOwn+intercept))
    +print(f"MSE with Sklearn intercept")
    +print(MSE(y,ypredictSKL))
    +
    +plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
    +plt.plot(x, skl.predict(X), "--", label="Sklearn (fit_intercept=True)")
    +plt.grid()
    +plt.legend()
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). +

    + +

    Printing the MSE, we see first that both methods give the same MSE, as +they should. However, when we move to for example Ridge regression, +the way we treat the intercept may give a larger or smaller MSE, +meaning that the MSE can be penalized by the value of the +intercept. Not including the intercept in the fit, means that the +regularization term does not include \( \beta_0 \). For different values +of \( \lambda \), this may lead to different MSE values. +

    + +

    To remind the reader, the regularization term, with the intercept in Ridge regression, is given by

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2, +$$ + +

    but when we take out the intercept, this equation becomes

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2. +$$ + +

    For Lasso regression we have

    +$$ +\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert. +$$ + +

    It means that, when scaling the design matrix and the outputs/targets, +by subtracting the mean values, we have an optimization problem which +is not penalized by the intercept. The MSE value can then be smaller +since it focuses only on the remaining quantities. If we however bring +back the intercept, we will get a MSE which then contains the +intercept. +

    + +

    Armed with this wisdom, we attempt first to simply set the intercept equal to False in our implementation of Ridge regression for our well-known vanilla data set.

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +
    +
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(3155)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree))
    +#We include explicitely the intercept column
    +for degree in range(Maxpolydegree):
    +    X[:,degree] = x**degree
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +p = Maxpolydegree
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
    +    # Note: we include the intercept column and no scaling
    +    RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
    +    RegRidge.fit(X_train,y_train)
    +    # and then make the prediction
    +    ytildeOwnRidge = X_train @ OwnRidgeBeta
    +    ypredictOwnRidge = X_test @ OwnRidgeBeta
    +    ytildeRidge = RegRidge.predict(X_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta)
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')
    +
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. +We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix. +What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering. +

    + + + +
    +
    +
    +
    +
    +
    import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.model_selection import train_test_split
    +from sklearn import linear_model
    +from sklearn.preprocessing import StandardScaler
    +
    +def MSE(y_data,y_model):
    +    n = np.size(y_model)
    +    return np.sum((y_data-y_model)**2)/n
    +# A seed just to ensure that the random numbers are the same for every run.
    +# Useful for eventual debugging.
    +np.random.seed(315)
    +
    +n = 100
    +x = np.random.rand(n)
    +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +
    +Maxpolydegree = 20
    +X = np.zeros((n,Maxpolydegree-1))
    +
    +for degree in range(1,Maxpolydegree): #No intercept column
    +    X[:,degree-1] = x**(degree)
    +
    +# We split the data in test and training data
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
    +X_train_mean = np.mean(X_train,axis=0)
    +#Center by removing mean from each feature
    +X_train_scaled = X_train - X_train_mean 
    +X_test_scaled = X_test - X_train_mean
    +#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
    +#Remove the intercept from the training data.
    +y_scaler = np.mean(y_train)           
    +y_train_scaled = y_train - y_scaler   
    +
    +p = Maxpolydegree-1
    +I = np.eye(p,p)
    +# Decide which values of lambda to use
    +nlambdas = 6
    +MSEOwnRidgePredict = np.zeros(nlambdas)
    +MSERidgePredict = np.zeros(nlambdas)
    +
    +lambdas = np.logspace(-4, 2, nlambdas)
    +for i in range(nlambdas):
    +    lmb = lambdas[i]
    +    OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
    +    intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
    +    #Add intercept to prediction
    +    ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler 
    +    RegRidge = linear_model.Ridge(lmb)
    +    RegRidge.fit(X_train,y_train)
    +    ypredictRidge = RegRidge.predict(X_test)
    +    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    +    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    +    print("Beta values for own Ridge implementation")
    +    print(OwnRidgeBeta) #Intercept is given by mean of target variable
    +    print("Beta values for Scikit-Learn Ridge implementation")
    +    print(RegRidge.coef_)
    +    print('Intercept from own implementation:')
    +    print(intercept_)
    +    print('Intercept from Scikit-Learn Ridge implementation')
    +    print(RegRidge.intercept_)
    +    print("MSE values for own Ridge implementation")
    +    print(MSEOwnRidgePredict[i])
    +    print("MSE values for Scikit-Learn Ridge implementation")
    +    print(MSERidgePredict[i])
    +
    +
    +# Now plot the results
    +plt.figure()
    +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
    +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
    +plt.xlabel('log10(lambda)')
    +plt.ylabel('MSE')
    +plt.legend()
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +

    We see here, when compared to the code which includes explicitely the +intercept column, that our MSE value is actually smaller. This is +because the regularization term does not include the intercept value +\( \beta_0 \) in the fitting. This applies to Lasso regularization as +well. It means that our optimization is now done only with the +centered matrix and/or vector that enter the fitting procedure. +

    +
    © 1999-2025, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license diff --git a/doc/pub/week35/ipynb/ipynb-week35-src.tar.gz b/doc/pub/week35/ipynb/ipynb-week35-src.tar.gz index b58ffd77726cc687a57e338474d6c2ac5254c6d1..525c7d87dc697a523c4f6dc658dc9767b1b8f39d 100644 GIT binary patch delta 170 zcmV;b09F6L0l)zzABzY8**~pl00ZsM%?iRW3t{UA^SrM;033r1h5!fv0NEi(w*UYD delta 169 zcmV;a09OCN0lxtyABzY8MRBZX00ZsM%?iRW3+Ocu`0vwc?ksHTq5ad0%?~*QzI&00;m8Xnahu diff --git a/doc/pub/week35/ipynb/week35.ipynb b/doc/pub/week35/ipynb/week35.ipynb index 206156855..b5da95d0a 100644 --- a/doc/pub/week35/ipynb/week35.ipynb +++ b/doc/pub/week35/ipynb/week35.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "1660940a", + "id": "d3666266", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "6c80589e", + "id": "f9417425", "metadata": { "editable": true }, @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "619797e6", + "id": "aaba55da", "metadata": { "editable": true }, @@ -49,7 +49,7 @@ }, { "cell_type": "markdown", - "id": "3abb04ff", + "id": "c4e2f992", "metadata": { "editable": true }, @@ -71,7 +71,7 @@ }, { "cell_type": "markdown", - "id": "6ec4becf", + "id": "f0040522", "metadata": { "editable": true }, @@ -104,7 +104,7 @@ }, { "cell_type": "markdown", - "id": "68f8d1dd", + "id": "d64fcc78", "metadata": { "editable": true }, @@ -120,7 +120,7 @@ }, { "cell_type": "markdown", - "id": "aa8fabf8", + "id": "a1ada6ce", "metadata": { "editable": true }, @@ -132,7 +132,7 @@ }, { "cell_type": "markdown", - "id": "8d9e9cc1", + "id": "5645817b", "metadata": { "editable": true }, @@ -142,7 +142,7 @@ }, { "cell_type": "markdown", - "id": "d99ff827", + "id": "30bee0bb", "metadata": { "editable": true }, @@ -154,7 +154,7 @@ }, { "cell_type": "markdown", - "id": "0fc8acad", + "id": "e6e20e11", "metadata": { "editable": true }, @@ -175,7 +175,7 @@ }, { "cell_type": "markdown", - "id": "6fbfbdae", + "id": "7bfd72c1", "metadata": { "editable": true }, @@ -187,7 +187,7 @@ }, { "cell_type": "markdown", - "id": "0042f819", + "id": "69275ea0", "metadata": { "editable": true }, @@ -200,7 +200,7 @@ }, { "cell_type": "markdown", - "id": "0adc0559", + "id": "976bb1cb", "metadata": { "editable": true }, @@ -212,7 +212,7 @@ }, { "cell_type": "markdown", - "id": "2ca124d9", + "id": "0d85580a", "metadata": { "editable": true }, @@ -224,7 +224,7 @@ }, { "cell_type": "markdown", - "id": "95c36c7f", + "id": "5a8ac559", "metadata": { "editable": true }, @@ -234,7 +234,7 @@ }, { "cell_type": "markdown", - "id": "6c9de3bb", + "id": "64f74e34", "metadata": { "editable": true }, @@ -246,7 +246,7 @@ }, { "cell_type": "markdown", - "id": "ebd13cde", + "id": "8d503698", "metadata": { "editable": true }, @@ -259,7 +259,7 @@ }, { "cell_type": "markdown", - "id": "467ef180", + "id": "787c6923", "metadata": { "editable": true }, @@ -271,7 +271,7 @@ }, { "cell_type": "markdown", - "id": "0c244885", + "id": "54c2b6dc", "metadata": { "editable": true }, @@ -281,7 +281,7 @@ }, { "cell_type": "markdown", - "id": "47eef2c3", + "id": "a8114ec8", "metadata": { "editable": true }, @@ -293,7 +293,7 @@ }, { "cell_type": "markdown", - "id": "cbf724f2", + "id": "650e695e", "metadata": { "editable": true }, @@ -305,7 +305,7 @@ }, { "cell_type": "markdown", - "id": "632c97b0", + "id": "885a3110", "metadata": { "editable": true }, @@ -316,7 +316,7 @@ }, { "cell_type": "markdown", - "id": "3633326b", + "id": "9fbaf50f", "metadata": { "editable": true }, @@ -328,7 +328,7 @@ }, { "cell_type": "markdown", - "id": "14c5dc89", + "id": "2376fa59", "metadata": { "editable": true }, @@ -347,7 +347,7 @@ }, { "cell_type": "markdown", - "id": "8b9d690d", + "id": "937df3c9", "metadata": { "editable": true }, @@ -360,7 +360,7 @@ }, { "cell_type": "markdown", - "id": "e5d540fb", + "id": "e066a9bc", "metadata": { "editable": true }, @@ -370,7 +370,7 @@ }, { "cell_type": "markdown", - "id": "a1064ef1", + "id": "af821847", "metadata": { "editable": true }, @@ -382,7 +382,7 @@ }, { "cell_type": "markdown", - "id": "8003adb3", + "id": "5e7ddc8b", "metadata": { "editable": true }, @@ -392,7 +392,7 @@ }, { "cell_type": "markdown", - "id": "d0052f64", + "id": "47d5910e", "metadata": { "editable": true }, @@ -404,7 +404,7 @@ }, { "cell_type": "markdown", - "id": "8f594583", + "id": "6d62a75a", "metadata": { "editable": true }, @@ -414,7 +414,7 @@ }, { "cell_type": "markdown", - "id": "5c429171", + "id": "58d10371", "metadata": { "editable": true }, @@ -426,7 +426,7 @@ }, { "cell_type": "markdown", - "id": "5e92187e", + "id": "1f349d3b", "metadata": { "editable": true }, @@ -437,7 +437,7 @@ }, { "cell_type": "markdown", - "id": "49ee1349", + "id": "14a30c64", "metadata": { "editable": true }, @@ -449,7 +449,7 @@ }, { "cell_type": "markdown", - "id": "7ad5ecf6", + "id": "b2faea7f", "metadata": { "editable": true }, @@ -459,7 +459,7 @@ }, { "cell_type": "markdown", - "id": "838e8942", + "id": "5810b814", "metadata": { "editable": true }, @@ -471,7 +471,7 @@ }, { "cell_type": "markdown", - "id": "56e39117", + "id": "fcf3c26e", "metadata": { "editable": true }, @@ -481,7 +481,7 @@ }, { "cell_type": "markdown", - "id": "6ce59cb2", + "id": "a4453721", "metadata": { "editable": true }, @@ -493,7 +493,7 @@ }, { "cell_type": "markdown", - "id": "f196d6b7", + "id": "38da802c", "metadata": { "editable": true }, @@ -513,7 +513,7 @@ }, { "cell_type": "markdown", - "id": "adfed9eb", + "id": "bdc1020c", "metadata": { "editable": true }, @@ -540,7 +540,7 @@ }, { "cell_type": "markdown", - "id": "b06d3aa7", + "id": "c69738c8", "metadata": { "editable": true }, @@ -552,7 +552,7 @@ }, { "cell_type": "markdown", - "id": "36a124c7", + "id": "c180c4b8", "metadata": { "editable": true }, @@ -564,7 +564,7 @@ }, { "cell_type": "markdown", - "id": "2dd13864", + "id": "e273b465", "metadata": { "editable": true }, @@ -580,7 +580,7 @@ }, { "cell_type": "markdown", - "id": "84b39d01", + "id": "89031202", "metadata": { "editable": true }, @@ -598,7 +598,7 @@ }, { "cell_type": "markdown", - "id": "8375a5f1", + "id": "a5bfe37a", "metadata": { "editable": true }, @@ -610,7 +610,7 @@ }, { "cell_type": "markdown", - "id": "9fe2c198", + "id": "9fe43da2", "metadata": { "editable": true }, @@ -622,7 +622,7 @@ }, { "cell_type": "markdown", - "id": "42f9a620", + "id": "17f42b20", "metadata": { "editable": true }, @@ -633,7 +633,7 @@ }, { "cell_type": "markdown", - "id": "be600b2f", + "id": "6e12c604", "metadata": { "editable": true }, @@ -645,7 +645,7 @@ }, { "cell_type": "markdown", - "id": "610790a0", + "id": "c03524f3", "metadata": { "editable": true }, @@ -655,7 +655,7 @@ }, { "cell_type": "markdown", - "id": "2f4f16f6", + "id": "66768181", "metadata": { "editable": true }, @@ -667,7 +667,7 @@ }, { "cell_type": "markdown", - "id": "63efa745", + "id": "23864bbb", "metadata": { "editable": true }, @@ -681,7 +681,7 @@ }, { "cell_type": "markdown", - "id": "80846216", + "id": "c39fc9ec", "metadata": { "editable": true }, @@ -693,7 +693,7 @@ }, { "cell_type": "markdown", - "id": "5366b7f0", + "id": "9733262a", "metadata": { "editable": true }, @@ -705,7 +705,7 @@ }, { "cell_type": "markdown", - "id": "f90c5c2e", + "id": "dc160822", "metadata": { "editable": true }, @@ -717,7 +717,7 @@ }, { "cell_type": "markdown", - "id": "bcbf16d2", + "id": "1211e28f", "metadata": { "editable": true }, @@ -727,7 +727,7 @@ }, { "cell_type": "markdown", - "id": "8f6c9660", + "id": "939dfa18", "metadata": { "editable": true }, @@ -739,7 +739,7 @@ }, { "cell_type": "markdown", - "id": "b71835a2", + "id": "f1d30438", "metadata": { "editable": true }, @@ -751,7 +751,7 @@ }, { "cell_type": "markdown", - "id": "1d6f47e4", + "id": "bdb64e7a", "metadata": { "editable": true }, @@ -763,7 +763,7 @@ }, { "cell_type": "markdown", - "id": "6afe8447", + "id": "4952d52d", "metadata": { "editable": true }, @@ -777,7 +777,7 @@ }, { "cell_type": "markdown", - "id": "c7f905a9", + "id": "f7ea9937", "metadata": { "editable": true }, @@ -789,7 +789,7 @@ }, { "cell_type": "markdown", - "id": "71cf59c4", + "id": "31e1ef19", "metadata": { "editable": true }, @@ -801,7 +801,7 @@ }, { "cell_type": "markdown", - "id": "ad69e6c4", + "id": "22628265", "metadata": { "editable": true }, @@ -813,7 +813,7 @@ }, { "cell_type": "markdown", - "id": "0ec57671", + "id": "d2e9c679", "metadata": { "editable": true }, @@ -823,7 +823,7 @@ }, { "cell_type": "markdown", - "id": "763ea011", + "id": "c03929d2", "metadata": { "editable": true }, @@ -835,7 +835,7 @@ }, { "cell_type": "markdown", - "id": "c02d1744", + "id": "131c4d51", "metadata": { "editable": true }, @@ -845,7 +845,7 @@ }, { "cell_type": "markdown", - "id": "04fabf5b", + "id": "27cc18f9", "metadata": { "editable": true }, @@ -857,7 +857,7 @@ }, { "cell_type": "markdown", - "id": "1197c6e5", + "id": "f78fde44", "metadata": { "editable": true }, @@ -867,7 +867,7 @@ }, { "cell_type": "markdown", - "id": "3e3abc5e", + "id": "b3117122", "metadata": { "editable": true }, @@ -879,7 +879,7 @@ }, { "cell_type": "markdown", - "id": "e92f0ed8", + "id": "6025f5d9", "metadata": { "editable": true }, @@ -891,7 +891,7 @@ }, { "cell_type": "markdown", - "id": "4920f058", + "id": "f6d884e9", "metadata": { "editable": true }, @@ -903,7 +903,7 @@ }, { "cell_type": "markdown", - "id": "44745024", + "id": "d7241fbc", "metadata": { "editable": true }, @@ -918,7 +918,7 @@ }, { "cell_type": "markdown", - "id": "de3c1376", + "id": "30ea6892", "metadata": { "editable": true }, @@ -930,7 +930,7 @@ }, { "cell_type": "markdown", - "id": "4e6d099b", + "id": "ada2be4f", "metadata": { "editable": true }, @@ -940,7 +940,7 @@ }, { "cell_type": "markdown", - "id": "14e5046c", + "id": "131da9a2", "metadata": { "editable": true }, @@ -952,7 +952,7 @@ }, { "cell_type": "markdown", - "id": "ce2b1125", + "id": "15d00d29", "metadata": { "editable": true }, @@ -962,7 +962,7 @@ }, { "cell_type": "markdown", - "id": "b9e66e41", + "id": "bd4077dd", "metadata": { "editable": true }, @@ -974,7 +974,7 @@ }, { "cell_type": "markdown", - "id": "fefc6fd2", + "id": "a0fd8aac", "metadata": { "editable": true }, @@ -984,7 +984,7 @@ }, { "cell_type": "markdown", - "id": "788487c4", + "id": "ae370d67", "metadata": { "editable": true }, @@ -996,7 +996,7 @@ }, { "cell_type": "markdown", - "id": "a08b45ce", + "id": "687aa950", "metadata": { "editable": true }, @@ -1008,7 +1008,7 @@ }, { "cell_type": "markdown", - "id": "0f822af8", + "id": "4159fec2", "metadata": { "editable": true }, @@ -1020,7 +1020,7 @@ }, { "cell_type": "markdown", - "id": "b617b65d", + "id": "e5f8e212", "metadata": { "editable": true }, @@ -1030,7 +1030,7 @@ }, { "cell_type": "markdown", - "id": "ac8432a4", + "id": "c26a9c8f", "metadata": { "editable": true }, @@ -1042,7 +1042,7 @@ }, { "cell_type": "markdown", - "id": "cb322b7c", + "id": "859c590b", "metadata": { "editable": true }, @@ -1055,7 +1055,7 @@ }, { "cell_type": "markdown", - "id": "4c65add0", + "id": "fbb8dd7f", "metadata": { "editable": true }, @@ -1067,7 +1067,7 @@ }, { "cell_type": "markdown", - "id": "cf4a0f92", + "id": "bf096607", "metadata": { "editable": true }, @@ -1077,7 +1077,7 @@ }, { "cell_type": "markdown", - "id": "965b48e0", + "id": "428e2635", "metadata": { "editable": true }, @@ -1089,7 +1089,7 @@ }, { "cell_type": "markdown", - "id": "06eb8bc2", + "id": "81ff9c77", "metadata": { "editable": true }, @@ -1099,7 +1099,7 @@ }, { "cell_type": "markdown", - "id": "99f96df5", + "id": "a4b9ca1c", "metadata": { "editable": true }, @@ -1111,7 +1111,7 @@ }, { "cell_type": "markdown", - "id": "b623b7b6", + "id": "b64449f2", "metadata": { "editable": true }, @@ -1121,7 +1121,7 @@ }, { "cell_type": "markdown", - "id": "63517497", + "id": "6a104348", "metadata": { "editable": true }, @@ -1133,7 +1133,7 @@ }, { "cell_type": "markdown", - "id": "ecf6bfde", + "id": "ce07e978", "metadata": { "editable": true }, @@ -1143,7 +1143,7 @@ }, { "cell_type": "markdown", - "id": "cad8192e", + "id": "f81215cb", "metadata": { "editable": true }, @@ -1155,7 +1155,7 @@ }, { "cell_type": "markdown", - "id": "815292b1", + "id": "54e70573", "metadata": { "editable": true }, @@ -1165,7 +1165,7 @@ }, { "cell_type": "markdown", - "id": "04845862", + "id": "2f2f537a", "metadata": { "editable": true }, @@ -1177,7 +1177,7 @@ }, { "cell_type": "markdown", - "id": "51ba343e", + "id": "9fb9df02", "metadata": { "editable": true }, @@ -1193,7 +1193,7 @@ }, { "cell_type": "markdown", - "id": "3fa42398", + "id": "2dff6039", "metadata": { "editable": true }, @@ -1205,7 +1205,7 @@ }, { "cell_type": "markdown", - "id": "d036b723", + "id": "783debc8", "metadata": { "editable": true }, @@ -1215,7 +1215,7 @@ }, { "cell_type": "markdown", - "id": "a0845534", + "id": "42965f7e", "metadata": { "editable": true }, @@ -1227,7 +1227,7 @@ }, { "cell_type": "markdown", - "id": "a9b057c9", + "id": "57e8d8fe", "metadata": { "editable": true }, @@ -1245,7 +1245,7 @@ }, { "cell_type": "markdown", - "id": "f85865c7", + "id": "c43b98ff", "metadata": { "editable": true }, @@ -1257,7 +1257,7 @@ }, { "cell_type": "markdown", - "id": "2aeaa95f", + "id": "6ff581de", "metadata": { "editable": true }, @@ -1269,7 +1269,7 @@ }, { "cell_type": "markdown", - "id": "422cbde8", + "id": "8f4462e2", "metadata": { "editable": true }, @@ -1279,7 +1279,7 @@ }, { "cell_type": "markdown", - "id": "9da44d74", + "id": "c935b169", "metadata": { "editable": true }, @@ -1291,7 +1291,7 @@ }, { "cell_type": "markdown", - "id": "2ae02ff4", + "id": "50b87db5", "metadata": { "editable": true }, @@ -1301,7 +1301,7 @@ }, { "cell_type": "markdown", - "id": "747afada", + "id": "2599f182", "metadata": { "editable": true }, @@ -1313,7 +1313,7 @@ }, { "cell_type": "markdown", - "id": "bafb855e", + "id": "6c07218b", "metadata": { "editable": true }, @@ -1323,7 +1323,7 @@ }, { "cell_type": "markdown", - "id": "10500f00", + "id": "d1058185", "metadata": { "editable": true }, @@ -1337,7 +1337,7 @@ }, { "cell_type": "markdown", - "id": "51c2bd12", + "id": "951d1097", "metadata": { "editable": true }, @@ -1349,7 +1349,7 @@ }, { "cell_type": "markdown", - "id": "38728a85", + "id": "73e7100b", "metadata": { "editable": true }, @@ -1360,7 +1360,7 @@ }, { "cell_type": "markdown", - "id": "ca27d043", + "id": "b7caf7f6", "metadata": { "editable": true }, @@ -1373,7 +1373,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "a156aead", + "id": "01229cfb", "metadata": { "collapsed": false, "editable": true @@ -1400,7 +1400,7 @@ }, { "cell_type": "markdown", - "id": "b1884daf", + "id": "f4a3896e", "metadata": { "editable": true }, @@ -1411,7 +1411,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "0876de90", + "id": "6ca9744d", "metadata": { "collapsed": false, "editable": true @@ -1424,7 +1424,7 @@ }, { "cell_type": "markdown", - "id": "c2cd9d7d", + "id": "c434f72f", "metadata": { "editable": true }, @@ -1438,7 +1438,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "c7ec7d38", + "id": "d4131dbc", "metadata": { "collapsed": false, "editable": true @@ -1451,7 +1451,7 @@ }, { "cell_type": "markdown", - "id": "4eb63b06", + "id": "e5cff99e", "metadata": { "editable": true }, @@ -1462,7 +1462,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "e6e5c603", + "id": "c1b15227", "metadata": { "collapsed": false, "editable": true @@ -1474,7 +1474,7 @@ }, { "cell_type": "markdown", - "id": "04d4fe67", + "id": "a0a432b4", "metadata": { "editable": true }, @@ -1485,7 +1485,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "ebe13c98", + "id": "319c1a33", "metadata": { "collapsed": false, "editable": true @@ -1501,7 +1501,7 @@ }, { "cell_type": "markdown", - "id": "b362e065", + "id": "98d492e3", "metadata": { "editable": true }, @@ -1512,7 +1512,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "04998ff6", + "id": "52ad17b3", "metadata": { "collapsed": false, "editable": true @@ -1526,7 +1526,7 @@ }, { "cell_type": "markdown", - "id": "3ce03a03", + "id": "29abfbb7", "metadata": { "editable": true }, @@ -1547,7 +1547,7 @@ }, { "cell_type": "markdown", - "id": "02895269", + "id": "503f0621", "metadata": { "editable": true }, @@ -1558,7 +1558,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "72b8f85c", + "id": "6927de82", "metadata": { "collapsed": false, "editable": true @@ -1611,7 +1611,7 @@ }, { "cell_type": "markdown", - "id": "8ad305fe", + "id": "0b4f15f1", "metadata": { "editable": true }, @@ -1622,7 +1622,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "6e34cfeb", + "id": "ac968b0f", "metadata": { "collapsed": false, "editable": true @@ -1647,7 +1647,7 @@ }, { "cell_type": "markdown", - "id": "b04b30e4", + "id": "4ea3a52b", "metadata": { "editable": true }, @@ -1659,7 +1659,7 @@ }, { "cell_type": "markdown", - "id": "a24d1b04", + "id": "a027a9a5", "metadata": { "editable": true }, @@ -1688,7 +1688,7 @@ }, { "cell_type": "markdown", - "id": "080c9be7", + "id": "b0b3e66c", "metadata": { "editable": true }, @@ -1713,7 +1713,7 @@ }, { "cell_type": "markdown", - "id": "44a0ad2f", + "id": "8af30d8b", "metadata": { "editable": true }, @@ -1733,7 +1733,7 @@ }, { "cell_type": "markdown", - "id": "7b4e35e2", + "id": "cd639711", "metadata": { "editable": true }, @@ -1760,7 +1760,7 @@ }, { "cell_type": "markdown", - "id": "8b81901f", + "id": "efb38ba0", "metadata": { "editable": true }, @@ -1773,7 +1773,7 @@ }, { "cell_type": "markdown", - "id": "8a0c0ed1", + "id": "bcc3ee33", "metadata": { "editable": true }, @@ -1785,7 +1785,7 @@ }, { "cell_type": "markdown", - "id": "d7d35571", + "id": "0c4c440b", "metadata": { "editable": true }, @@ -1796,7 +1796,7 @@ }, { "cell_type": "markdown", - "id": "6719e2c5", + "id": "b006a855", "metadata": { "editable": true }, @@ -1812,7 +1812,7 @@ { "cell_type": "code", "execution_count": 9, - "id": "cedbcc2d", + "id": "98079995", "metadata": { "collapsed": false, "editable": true @@ -1846,7 +1846,7 @@ }, { "cell_type": "markdown", - "id": "9fcfc7f6", + "id": "4d8bc84b", "metadata": { "editable": true }, @@ -1856,7 +1856,7 @@ }, { "cell_type": "markdown", - "id": "17677144", + "id": "78d54efa", "metadata": { "editable": true }, @@ -1871,7 +1871,7 @@ }, { "cell_type": "markdown", - "id": "c0c0d7a8", + "id": "2c6126fc", "metadata": { "editable": true }, @@ -1883,7 +1883,7 @@ }, { "cell_type": "markdown", - "id": "82d52b85", + "id": "2cc97a3e", "metadata": { "editable": true }, @@ -1893,7 +1893,7 @@ }, { "cell_type": "markdown", - "id": "a1ea3354", + "id": "9e0ea279", "metadata": { "editable": true }, @@ -1909,7 +1909,7 @@ { "cell_type": "code", "execution_count": 10, - "id": "a06e8556", + "id": "d00a25ea", "metadata": { "collapsed": false, "editable": true @@ -1926,7 +1926,7 @@ }, { "cell_type": "markdown", - "id": "848e7706", + "id": "864015a1", "metadata": { "editable": true }, @@ -1939,7 +1939,7 @@ { "cell_type": "code", "execution_count": 11, - "id": "5e87326f", + "id": "5ba30f30", "metadata": { "collapsed": false, "editable": true @@ -1986,7 +1986,7 @@ }, { "cell_type": "markdown", - "id": "6c359908", + "id": "8dc3cb9f", "metadata": { "editable": true }, @@ -2000,7 +2000,7 @@ }, { "cell_type": "markdown", - "id": "3806b6d7", + "id": "068d5e0a", "metadata": { "editable": true }, @@ -2012,7 +2012,7 @@ }, { "cell_type": "markdown", - "id": "3fbe7b9f", + "id": "6feb2964", "metadata": { "editable": true }, @@ -2024,7 +2024,7 @@ }, { "cell_type": "markdown", - "id": "07d065b1", + "id": "df3d9523", "metadata": { "editable": true }, @@ -2036,7 +2036,7 @@ }, { "cell_type": "markdown", - "id": "7594c295", + "id": "7e8785d1", "metadata": { "editable": true }, @@ -2046,7 +2046,7 @@ }, { "cell_type": "markdown", - "id": "0072ac1a", + "id": "ef3be4f1", "metadata": { "editable": true }, @@ -2058,7 +2058,7 @@ }, { "cell_type": "markdown", - "id": "c06a93bc", + "id": "3dcd8cf6", "metadata": { "editable": true }, @@ -2068,7 +2068,7 @@ }, { "cell_type": "markdown", - "id": "cbd67a39", + "id": "7625f933", "metadata": { "editable": true }, @@ -2080,7 +2080,7 @@ }, { "cell_type": "markdown", - "id": "06e8a8ce", + "id": "ef6f7710", "metadata": { "editable": true }, @@ -2091,7 +2091,7 @@ }, { "cell_type": "markdown", - "id": "44f0cf2a", + "id": "b52129d8", "metadata": { "editable": true }, @@ -2103,7 +2103,7 @@ }, { "cell_type": "markdown", - "id": "53c56046", + "id": "c0670f24", "metadata": { "editable": true }, @@ -2115,7 +2115,7 @@ }, { "cell_type": "markdown", - "id": "ae2f06b1", + "id": "a30df157", "metadata": { "editable": true }, @@ -2125,7 +2125,7 @@ }, { "cell_type": "markdown", - "id": "aa6d685f", + "id": "75243d64", "metadata": { "editable": true }, @@ -2137,7 +2137,7 @@ }, { "cell_type": "markdown", - "id": "1f804e4c", + "id": "9c359540", "metadata": { "editable": true }, @@ -2149,7 +2149,7 @@ }, { "cell_type": "markdown", - "id": "b81d6ca7", + "id": "fa82ae4e", "metadata": { "editable": true }, @@ -2159,7 +2159,7 @@ }, { "cell_type": "markdown", - "id": "8c9f9694", + "id": "65734ef1", "metadata": { "editable": true }, @@ -2171,7 +2171,7 @@ }, { "cell_type": "markdown", - "id": "c5dd165e", + "id": "650117d0", "metadata": { "editable": true }, @@ -2181,7 +2181,7 @@ }, { "cell_type": "markdown", - "id": "c85f8ef6", + "id": "06771b2d", "metadata": { "editable": true }, @@ -2193,7 +2193,7 @@ }, { "cell_type": "markdown", - "id": "74f30d7b", + "id": "690e56c7", "metadata": { "editable": true }, @@ -2203,7 +2203,7 @@ }, { "cell_type": "markdown", - "id": "75c5c93f", + "id": "db5e7402", "metadata": { "editable": true }, @@ -2243,7 +2243,7 @@ }, { "cell_type": "markdown", - "id": "0cb6452e", + "id": "a7813ad4", "metadata": { "editable": true }, @@ -2260,7 +2260,7 @@ }, { "cell_type": "markdown", - "id": "3af43196", + "id": "2108743e", "metadata": { "editable": true }, @@ -2283,7 +2283,7 @@ }, { "cell_type": "markdown", - "id": "470b331f", + "id": "3f63b5de", "metadata": { "editable": true }, @@ -2300,7 +2300,7 @@ }, { "cell_type": "markdown", - "id": "d357c80c", + "id": "05003c17", "metadata": { "editable": true }, @@ -2319,7 +2319,7 @@ }, { "cell_type": "markdown", - "id": "06c5d53f", + "id": "315a72a0", "metadata": { "editable": true }, @@ -2330,7 +2330,7 @@ }, { "cell_type": "markdown", - "id": "f791a2a3", + "id": "66ceac64", "metadata": { "editable": true }, @@ -2342,7 +2342,7 @@ }, { "cell_type": "markdown", - "id": "1ed601ad", + "id": "35b697b9", "metadata": { "editable": true }, @@ -2360,7 +2360,7 @@ }, { "cell_type": "markdown", - "id": "5e84f076", + "id": "4712a94c", "metadata": { "editable": true }, @@ -2376,7 +2376,7 @@ }, { "cell_type": "markdown", - "id": "5b04faea", + "id": "c11c7415", "metadata": { "editable": true }, @@ -2388,7 +2388,7 @@ }, { "cell_type": "markdown", - "id": "5545e598", + "id": "7eecbc75", "metadata": { "editable": true }, @@ -2398,7 +2398,7 @@ }, { "cell_type": "markdown", - "id": "9e783636", + "id": "3577f76a", "metadata": { "editable": true }, @@ -2411,7 +2411,7 @@ }, { "cell_type": "markdown", - "id": "0ac94e0f", + "id": "e1cad910", "metadata": { "editable": true }, @@ -2423,7 +2423,7 @@ }, { "cell_type": "markdown", - "id": "b5c15c50", + "id": "e1ed7f36", "metadata": { "editable": true }, @@ -2433,7 +2433,7 @@ }, { "cell_type": "markdown", - "id": "cc471bc5", + "id": "d9797b9a", "metadata": { "editable": true }, @@ -2446,7 +2446,7 @@ }, { "cell_type": "markdown", - "id": "82defafc", + "id": "8a7d60f5", "metadata": { "editable": true }, @@ -2456,7 +2456,7 @@ }, { "cell_type": "markdown", - "id": "01e8467e", + "id": "597db6eb", "metadata": { "editable": true }, @@ -2468,7 +2468,7 @@ }, { "cell_type": "markdown", - "id": "8e43e611", + "id": "03d92296", "metadata": { "editable": true }, @@ -2481,7 +2481,7 @@ }, { "cell_type": "markdown", - "id": "c86047ea", + "id": "135a3e67", "metadata": { "editable": true }, @@ -2494,7 +2494,7 @@ }, { "cell_type": "markdown", - "id": "8ba52e56", + "id": "21a76d33", "metadata": { "editable": true }, @@ -2506,7 +2506,7 @@ }, { "cell_type": "markdown", - "id": "3ef174de", + "id": "b7797c08", "metadata": { "editable": true }, @@ -2518,7 +2518,7 @@ }, { "cell_type": "markdown", - "id": "59e5ef87", + "id": "bd4d1e02", "metadata": { "editable": true }, @@ -2528,7 +2528,7 @@ }, { "cell_type": "markdown", - "id": "4538f3b7", + "id": "b46ffa29", "metadata": { "editable": true }, @@ -2541,7 +2541,7 @@ }, { "cell_type": "markdown", - "id": "d8868bd3", + "id": "82955735", "metadata": { "editable": true }, @@ -2553,7 +2553,7 @@ }, { "cell_type": "markdown", - "id": "7f89217d", + "id": "a489bfcb", "metadata": { "editable": true }, @@ -2565,7 +2565,7 @@ }, { "cell_type": "markdown", - "id": "69198132", + "id": "b79dc723", "metadata": { "editable": true }, @@ -2577,7 +2577,7 @@ }, { "cell_type": "markdown", - "id": "6c79852a", + "id": "b4e4b19c", "metadata": { "editable": true }, @@ -2589,7 +2589,7 @@ }, { "cell_type": "markdown", - "id": "68ff78d6", + "id": "34dc0808", "metadata": { "editable": true }, @@ -2603,7 +2603,7 @@ }, { "cell_type": "markdown", - "id": "a4640da0", + "id": "9e053bfc", "metadata": { "editable": true }, @@ -2615,7 +2615,7 @@ }, { "cell_type": "markdown", - "id": "4ff9450a", + "id": "b8cb20a9", "metadata": { "editable": true }, @@ -2625,7 +2625,7 @@ }, { "cell_type": "markdown", - "id": "75897968", + "id": "d218f1b3", "metadata": { "editable": true }, @@ -2637,7 +2637,7 @@ }, { "cell_type": "markdown", - "id": "ec398d53", + "id": "dc3e144e", "metadata": { "editable": true }, @@ -2649,7 +2649,7 @@ }, { "cell_type": "markdown", - "id": "d08c5e80", + "id": "b9dcf0b9", "metadata": { "editable": true }, @@ -2661,7 +2661,7 @@ }, { "cell_type": "markdown", - "id": "c42c0dca", + "id": "a118bee6", "metadata": { "editable": true }, @@ -2673,7 +2673,7 @@ }, { "cell_type": "markdown", - "id": "581ce12e", + "id": "a18637c1", "metadata": { "editable": true }, @@ -2685,7 +2685,7 @@ }, { "cell_type": "markdown", - "id": "11c8e658", + "id": "22ecf42d", "metadata": { "editable": true }, @@ -2708,7 +2708,7 @@ { "cell_type": "code", "execution_count": 12, - "id": "f1d02257", + "id": "2c933b95", "metadata": { "collapsed": false, "editable": true @@ -2784,7 +2784,7 @@ }, { "cell_type": "markdown", - "id": "16061bb5", + "id": "27fcd2cd", "metadata": { "editable": true }, @@ -2796,7 +2796,7 @@ }, { "cell_type": "markdown", - "id": "2c383a85", + "id": "fb501ab3", "metadata": { "editable": true }, @@ -2811,7 +2811,7 @@ }, { "cell_type": "markdown", - "id": "fb9dc580", + "id": "bbdc8c03", "metadata": { "editable": true }, @@ -2823,7 +2823,7 @@ }, { "cell_type": "markdown", - "id": "2b51ac49", + "id": "4f4e1b19", "metadata": { "editable": true }, @@ -2833,7 +2833,7 @@ }, { "cell_type": "markdown", - "id": "a0607286", + "id": "4b53a433", "metadata": { "editable": true }, @@ -2845,7 +2845,7 @@ }, { "cell_type": "markdown", - "id": "124b991b", + "id": "243e598d", "metadata": { "editable": true }, @@ -2855,7 +2855,7 @@ }, { "cell_type": "markdown", - "id": "4243178e", + "id": "9e54e87c", "metadata": { "editable": true }, @@ -2867,7 +2867,7 @@ }, { "cell_type": "markdown", - "id": "07ccdd45", + "id": "ca9080f4", "metadata": { "editable": true }, @@ -2879,7 +2879,7 @@ }, { "cell_type": "markdown", - "id": "71a9791d", + "id": "c0da749a", "metadata": { "editable": true }, @@ -2894,7 +2894,7 @@ }, { "cell_type": "markdown", - "id": "9c670e40", + "id": "73106e44", "metadata": { "editable": true }, @@ -2905,7 +2905,7 @@ }, { "cell_type": "markdown", - "id": "ecc628ec", + "id": "47a5df25", "metadata": { "editable": true }, @@ -2925,7 +2925,7 @@ }, { "cell_type": "markdown", - "id": "db67f6ed", + "id": "f75c5f9d", "metadata": { "editable": true }, @@ -2937,7 +2937,7 @@ }, { "cell_type": "markdown", - "id": "759bc1c0", + "id": "ae49bbbe", "metadata": { "editable": true }, @@ -2947,7 +2947,7 @@ }, { "cell_type": "markdown", - "id": "e3c043d8", + "id": "9927bcbf", "metadata": { "editable": true }, @@ -2959,7 +2959,7 @@ }, { "cell_type": "markdown", - "id": "c978a21b", + "id": "557aa14a", "metadata": { "editable": true }, @@ -2988,7 +2988,7 @@ }, { "cell_type": "markdown", - "id": "55aca27e", + "id": "b77be05b", "metadata": { "editable": true }, @@ -3015,7 +3015,7 @@ }, { "cell_type": "markdown", - "id": "1cbbdd65", + "id": "dbca187b", "metadata": { "editable": true }, @@ -3026,7 +3026,7 @@ { "cell_type": "code", "execution_count": 13, - "id": "b0732353", + "id": "52ee2c0a", "metadata": { "collapsed": false, "editable": true @@ -3066,7 +3066,7 @@ }, { "cell_type": "markdown", - "id": "6a94fc58", + "id": "9585fb61", "metadata": { "editable": true }, @@ -3083,7 +3083,7 @@ }, { "cell_type": "markdown", - "id": "4278d912", + "id": "1820219c", "metadata": { "editable": true }, @@ -3106,7 +3106,7 @@ }, { "cell_type": "markdown", - "id": "6317fa84", + "id": "61e763b3", "metadata": { "editable": true }, @@ -3120,7 +3120,7 @@ }, { "cell_type": "markdown", - "id": "d8a37273", + "id": "d215271f", "metadata": { "editable": true }, @@ -3139,7 +3139,7 @@ }, { "cell_type": "markdown", - "id": "26b26711", + "id": "9e097ba9", "metadata": { "editable": true }, @@ -3149,7 +3149,7 @@ }, { "cell_type": "markdown", - "id": "e3e71a86", + "id": "c11e5ea0", "metadata": { "editable": true }, @@ -3161,7 +3161,7 @@ }, { "cell_type": "markdown", - "id": "e62c78ea", + "id": "d7a4b260", "metadata": { "editable": true }, @@ -3175,7 +3175,7 @@ }, { "cell_type": "markdown", - "id": "dcd523b7", + "id": "dde8684e", "metadata": { "editable": true }, @@ -3187,7 +3187,7 @@ }, { "cell_type": "markdown", - "id": "5eb0e81c", + "id": "1ee4ec03", "metadata": { "editable": true }, @@ -3197,7 +3197,7 @@ }, { "cell_type": "markdown", - "id": "12e5f11b", + "id": "5db5fd8b", "metadata": { "editable": true }, @@ -3209,7 +3209,7 @@ }, { "cell_type": "markdown", - "id": "09be425b", + "id": "315fa545", "metadata": { "editable": true }, @@ -3226,7 +3226,7 @@ }, { "cell_type": "markdown", - "id": "cb064e74", + "id": "669553b8", "metadata": { "editable": true }, @@ -3236,7 +3236,7 @@ }, { "cell_type": "markdown", - "id": "e762ea61", + "id": "a5c69189", "metadata": { "editable": true }, @@ -3252,7 +3252,7 @@ }, { "cell_type": "markdown", - "id": "6b9160a8", + "id": "c98988f9", "metadata": { "editable": true }, @@ -3262,7 +3262,7 @@ }, { "cell_type": "markdown", - "id": "7593594e", + "id": "ff274815", "metadata": { "editable": true }, @@ -3278,7 +3278,7 @@ }, { "cell_type": "markdown", - "id": "0ef745d5", + "id": "be481f90", "metadata": { "editable": true }, @@ -3288,7 +3288,7 @@ }, { "cell_type": "markdown", - "id": "46ff419d", + "id": "5389a4fd", "metadata": { "editable": true }, @@ -3304,7 +3304,7 @@ }, { "cell_type": "markdown", - "id": "7fbb6b26", + "id": "38c8b825", "metadata": { "editable": true }, @@ -3314,7 +3314,7 @@ }, { "cell_type": "markdown", - "id": "f4059c27", + "id": "854c2801", "metadata": { "editable": true }, @@ -3331,7 +3331,7 @@ }, { "cell_type": "markdown", - "id": "e17a9b2d", + "id": "77b24821", "metadata": { "editable": true }, @@ -3343,7 +3343,7 @@ }, { "cell_type": "markdown", - "id": "1be34c16", + "id": "9cb3e178", "metadata": { "editable": true }, @@ -3355,7 +3355,7 @@ }, { "cell_type": "markdown", - "id": "f418cc67", + "id": "a3e31623", "metadata": { "editable": true }, @@ -3367,7 +3367,7 @@ }, { "cell_type": "markdown", - "id": "737ffdbd", + "id": "28150e6b", "metadata": { "editable": true }, @@ -3377,7 +3377,7 @@ }, { "cell_type": "markdown", - "id": "4fd03e0b", + "id": "fc1f871e", "metadata": { "editable": true }, @@ -3389,7 +3389,7 @@ }, { "cell_type": "markdown", - "id": "53b1a1a7", + "id": "dd7f98b3", "metadata": { "editable": true }, @@ -3401,7 +3401,7 @@ }, { "cell_type": "markdown", - "id": "b7854b84", + "id": "b20a7cc6", "metadata": { "editable": true }, @@ -3413,7 +3413,7 @@ }, { "cell_type": "markdown", - "id": "055c9d11", + "id": "82d661e0", "metadata": { "editable": true }, @@ -3423,7 +3423,7 @@ }, { "cell_type": "markdown", - "id": "8f8dc5da", + "id": "705bb92a", "metadata": { "editable": true }, @@ -3435,7 +3435,7 @@ }, { "cell_type": "markdown", - "id": "13003e9e", + "id": "5c983fca", "metadata": { "editable": true }, @@ -3445,7 +3445,7 @@ }, { "cell_type": "markdown", - "id": "36baa1c1", + "id": "5b32695b", "metadata": { "editable": true }, @@ -3457,7 +3457,7 @@ }, { "cell_type": "markdown", - "id": "0a6568dd", + "id": "c4285bb3", "metadata": { "editable": true }, @@ -3474,7 +3474,7 @@ }, { "cell_type": "markdown", - "id": "aa0474a3", + "id": "1fa01f9a", "metadata": { "editable": true }, @@ -3486,7 +3486,7 @@ }, { "cell_type": "markdown", - "id": "d81e4d26", + "id": "c9d412c2", "metadata": { "editable": true }, @@ -3498,7 +3498,7 @@ }, { "cell_type": "markdown", - "id": "aace3412", + "id": "5307ebb8", "metadata": { "editable": true }, @@ -3508,7 +3508,7 @@ }, { "cell_type": "markdown", - "id": "f0cefdff", + "id": "ed2b38c3", "metadata": { "editable": true }, @@ -3520,7 +3520,7 @@ }, { "cell_type": "markdown", - "id": "77b3072c", + "id": "5dd9e137", "metadata": { "editable": true }, @@ -3531,7 +3531,7 @@ }, { "cell_type": "markdown", - "id": "c4fe1da3", + "id": "38a3902d", "metadata": { "editable": true }, @@ -3543,7 +3543,7 @@ }, { "cell_type": "markdown", - "id": "08e18c49", + "id": "53a81f9f", "metadata": { "editable": true }, @@ -3553,7 +3553,7 @@ }, { "cell_type": "markdown", - "id": "fb41b757", + "id": "c645a7d0", "metadata": { "editable": true }, @@ -3565,7 +3565,7 @@ }, { "cell_type": "markdown", - "id": "d5972c3f", + "id": "a3098a03", "metadata": { "editable": true }, @@ -3575,7 +3575,7 @@ }, { "cell_type": "markdown", - "id": "b042ca40", + "id": "08b90293", "metadata": { "editable": true }, @@ -3587,7 +3587,7 @@ }, { "cell_type": "markdown", - "id": "c50b8437", + "id": "1842a571", "metadata": { "editable": true }, @@ -3598,7 +3598,7 @@ }, { "cell_type": "markdown", - "id": "4a9a08da", + "id": "0290fcb9", "metadata": { "editable": true }, @@ -3610,7 +3610,7 @@ }, { "cell_type": "markdown", - "id": "a6cc43bc", + "id": "ecbc00d0", "metadata": { "editable": true }, @@ -3628,7 +3628,7 @@ }, { "cell_type": "markdown", - "id": "1140b18f", + "id": "670c8859", "metadata": { "editable": true }, @@ -3644,7 +3644,7 @@ }, { "cell_type": "markdown", - "id": "9b0ce6a4", + "id": "286fd62d", "metadata": { "editable": true }, @@ -3656,7 +3656,7 @@ }, { "cell_type": "markdown", - "id": "81d42d2e", + "id": "a51c433e", "metadata": { "editable": true }, @@ -3668,7 +3668,7 @@ }, { "cell_type": "markdown", - "id": "da0e1bbe", + "id": "d19e4786", "metadata": { "editable": true }, @@ -3680,7 +3680,7 @@ }, { "cell_type": "markdown", - "id": "43c2eefb", + "id": "56caedc2", "metadata": { "editable": true }, @@ -3693,7 +3693,7 @@ }, { "cell_type": "markdown", - "id": "f959c168", + "id": "d71c6ba0", "metadata": { "editable": true }, @@ -3709,7 +3709,7 @@ }, { "cell_type": "markdown", - "id": "b658ad6f", + "id": "3dfd726a", "metadata": { "editable": true }, @@ -3723,7 +3723,7 @@ }, { "cell_type": "markdown", - "id": "717ddf03", + "id": "5ed3d153", "metadata": { "editable": true }, @@ -3733,7 +3733,7 @@ }, { "cell_type": "markdown", - "id": "793b155d", + "id": "1ac50487", "metadata": { "editable": true }, @@ -3745,7 +3745,7 @@ }, { "cell_type": "markdown", - "id": "cb3fb411", + "id": "d3cc0098", "metadata": { "editable": true }, @@ -3755,7 +3755,7 @@ }, { "cell_type": "markdown", - "id": "c5ea6bd4", + "id": "d1adc309", "metadata": { "editable": true }, @@ -3767,7 +3767,7 @@ }, { "cell_type": "markdown", - "id": "8b1b2f04", + "id": "606b3621", "metadata": { "editable": true }, @@ -3777,7 +3777,7 @@ }, { "cell_type": "markdown", - "id": "6af257c8", + "id": "48dfd7a6", "metadata": { "editable": true }, @@ -3791,7 +3791,7 @@ }, { "cell_type": "markdown", - "id": "36458bd7", + "id": "f2d389fc", "metadata": { "editable": true }, @@ -3808,7 +3808,7 @@ }, { "cell_type": "markdown", - "id": "1fb74277", + "id": "93bee516", "metadata": { "editable": true }, @@ -3824,7 +3824,7 @@ }, { "cell_type": "markdown", - "id": "06acc410", + "id": "b19050a0", "metadata": { "editable": true }, @@ -3836,7 +3836,7 @@ }, { "cell_type": "markdown", - "id": "63631b7e", + "id": "d6f57474", "metadata": { "editable": true }, @@ -3849,7 +3849,7 @@ }, { "cell_type": "markdown", - "id": "122c2c56", + "id": "f7dd4e66", "metadata": { "editable": true }, @@ -3863,7 +3863,7 @@ }, { "cell_type": "markdown", - "id": "e2cb1169", + "id": "99d565fb", "metadata": { "editable": true }, @@ -3873,7 +3873,7 @@ }, { "cell_type": "markdown", - "id": "13fd6893", + "id": "c65756ab", "metadata": { "editable": true }, @@ -3886,7 +3886,7 @@ }, { "cell_type": "markdown", - "id": "b7fffc4e", + "id": "16c3efb7", "metadata": { "editable": true }, @@ -3905,7 +3905,7 @@ }, { "cell_type": "markdown", - "id": "dfd71bc3", + "id": "6e4194ff", "metadata": { "editable": true }, @@ -3917,7 +3917,7 @@ }, { "cell_type": "markdown", - "id": "736ba3eb", + "id": "5259c2eb", "metadata": { "editable": true }, @@ -3929,7 +3929,7 @@ }, { "cell_type": "markdown", - "id": "085291ee", + "id": "1b927e23", "metadata": { "editable": true }, @@ -3939,7 +3939,7 @@ }, { "cell_type": "markdown", - "id": "14097d0b", + "id": "34ab019a", "metadata": { "editable": true }, @@ -3951,7 +3951,7 @@ }, { "cell_type": "markdown", - "id": "3ebfdc21", + "id": "21075d57", "metadata": { "editable": true }, @@ -3964,7 +3964,7 @@ }, { "cell_type": "markdown", - "id": "7336aee6", + "id": "8833f450", "metadata": { "editable": true }, @@ -3983,7 +3983,7 @@ }, { "cell_type": "markdown", - "id": "9969b05b", + "id": "0813cc4c", "metadata": { "editable": true }, @@ -3993,7 +3993,7 @@ }, { "cell_type": "markdown", - "id": "9b6b8c4a", + "id": "50d7582d", "metadata": { "editable": true }, @@ -4012,7 +4012,7 @@ }, { "cell_type": "markdown", - "id": "df05134f", + "id": "8f85e67d", "metadata": { "editable": true }, @@ -4030,7 +4030,7 @@ }, { "cell_type": "markdown", - "id": "c093546b", + "id": "385c55d6", "metadata": { "editable": true }, @@ -4044,7 +4044,7 @@ }, { "cell_type": "markdown", - "id": "75a105dd", + "id": "ead461f7", "metadata": { "editable": true }, @@ -4059,7 +4059,7 @@ { "cell_type": "code", "execution_count": 14, - "id": "b2471ea4", + "id": "80c333dc", "metadata": { "collapsed": false, "editable": true @@ -4080,7 +4080,7 @@ }, { "cell_type": "markdown", - "id": "ff334dc3", + "id": "69f8b2df", "metadata": { "editable": true }, @@ -4097,7 +4097,7 @@ { "cell_type": "code", "execution_count": 15, - "id": "e209f8d6", + "id": "4a07353d", "metadata": { "collapsed": false, "editable": true @@ -4129,7 +4129,7 @@ }, { "cell_type": "markdown", - "id": "25263de6", + "id": "ee3bce85", "metadata": { "editable": true }, @@ -4143,7 +4143,7 @@ }, { "cell_type": "markdown", - "id": "049585bb", + "id": "3d15065a", "metadata": { "editable": true }, @@ -4156,7 +4156,7 @@ { "cell_type": "code", "execution_count": 16, - "id": "3ba71567", + "id": "488f60ab", "metadata": { "collapsed": false, "editable": true @@ -4181,7 +4181,7 @@ }, { "cell_type": "markdown", - "id": "c8c8e01d", + "id": "cafa6fa9", "metadata": { "editable": true }, @@ -4193,7 +4193,7 @@ }, { "cell_type": "markdown", - "id": "e4881f25", + "id": "ce70f3a8", "metadata": { "editable": true }, @@ -4205,7 +4205,7 @@ }, { "cell_type": "markdown", - "id": "7836e006", + "id": "10c922ba", "metadata": { "editable": true }, @@ -4215,7 +4215,7 @@ }, { "cell_type": "markdown", - "id": "25273c91", + "id": "0d791d47", "metadata": { "editable": true }, @@ -4232,7 +4232,7 @@ }, { "cell_type": "markdown", - "id": "d6936262", + "id": "533a270f", "metadata": { "editable": true }, @@ -4242,7 +4242,7 @@ }, { "cell_type": "markdown", - "id": "62adb1bf", + "id": "1b8bd026", "metadata": { "editable": true }, @@ -4257,7 +4257,7 @@ }, { "cell_type": "markdown", - "id": "53fd9388", + "id": "f3f09455", "metadata": { "editable": true }, @@ -4267,7 +4267,7 @@ }, { "cell_type": "markdown", - "id": "efba25c2", + "id": "bce601da", "metadata": { "editable": true }, @@ -4281,7 +4281,7 @@ }, { "cell_type": "markdown", - "id": "def71cfa", + "id": "85ed9781", "metadata": { "editable": true }, @@ -4293,7 +4293,7 @@ }, { "cell_type": "markdown", - "id": "6fe058cd", + "id": "7f6fcb9f", "metadata": { "editable": true }, @@ -4305,7 +4305,7 @@ }, { "cell_type": "markdown", - "id": "89169a7c", + "id": "6ea6f608", "metadata": { "editable": true }, @@ -4317,7 +4317,7 @@ }, { "cell_type": "markdown", - "id": "de471734", + "id": "005927d3", "metadata": { "editable": true }, @@ -4327,7 +4327,7 @@ }, { "cell_type": "markdown", - "id": "b2a55702", + "id": "c1f7c3dd", "metadata": { "editable": true }, @@ -4339,7 +4339,7 @@ }, { "cell_type": "markdown", - "id": "f096fd90", + "id": "a5c144b5", "metadata": { "editable": true }, @@ -4349,7 +4349,7 @@ }, { "cell_type": "markdown", - "id": "b0ed97a5", + "id": "73802678", "metadata": { "editable": true }, @@ -4366,7 +4366,7 @@ }, { "cell_type": "markdown", - "id": "7c473739", + "id": "aa888252", "metadata": { "editable": true }, @@ -4376,7 +4376,7 @@ }, { "cell_type": "markdown", - "id": "4209ea6b", + "id": "9d3363b0", "metadata": { "editable": true }, @@ -4388,7 +4388,7 @@ }, { "cell_type": "markdown", - "id": "43ea57c2", + "id": "49a1347a", "metadata": { "editable": true }, @@ -4398,7 +4398,7 @@ }, { "cell_type": "markdown", - "id": "b3c60d53", + "id": "1ba2b422", "metadata": { "editable": true }, @@ -4410,7 +4410,7 @@ }, { "cell_type": "markdown", - "id": "fe95894c", + "id": "5ffae339", "metadata": { "editable": true }, @@ -4424,7 +4424,7 @@ }, { "cell_type": "markdown", - "id": "b645c1d3", + "id": "cf729fa0", "metadata": { "editable": true }, @@ -4436,7 +4436,7 @@ }, { "cell_type": "markdown", - "id": "5b42f545", + "id": "57cf0b92", "metadata": { "editable": true }, @@ -4458,7 +4458,7 @@ }, { "cell_type": "markdown", - "id": "23cb329f", + "id": "c61e0046", "metadata": { "editable": true }, @@ -4470,7 +4470,7 @@ }, { "cell_type": "markdown", - "id": "5874588a", + "id": "66448cf4", "metadata": { "editable": true }, @@ -4485,7 +4485,7 @@ }, { "cell_type": "markdown", - "id": "5f13c123", + "id": "53d03309", "metadata": { "editable": true }, @@ -4497,7 +4497,7 @@ }, { "cell_type": "markdown", - "id": "cc0cc160", + "id": "530e38d2", "metadata": { "editable": true }, @@ -4509,7 +4509,7 @@ }, { "cell_type": "markdown", - "id": "95525455", + "id": "f073fbdc", "metadata": { "editable": true }, @@ -4519,7 +4519,7 @@ }, { "cell_type": "markdown", - "id": "20cb53cb", + "id": "cd18b54e", "metadata": { "editable": true }, @@ -4531,7 +4531,7 @@ }, { "cell_type": "markdown", - "id": "5ce5ef88", + "id": "c21ebf43", "metadata": { "editable": true }, @@ -4541,7 +4541,7 @@ }, { "cell_type": "markdown", - "id": "5e99eeac", + "id": "7b868e3b", "metadata": { "editable": true }, @@ -4553,7 +4553,7 @@ }, { "cell_type": "markdown", - "id": "e2b3e477", + "id": "e6e45cd5", "metadata": { "editable": true }, @@ -4563,7 +4563,7 @@ }, { "cell_type": "markdown", - "id": "2a5dbc01", + "id": "d54cddec", "metadata": { "editable": true }, @@ -4575,7 +4575,7 @@ }, { "cell_type": "markdown", - "id": "c29dc2fe", + "id": "3f887335", "metadata": { "editable": true }, @@ -4592,7 +4592,7 @@ }, { "cell_type": "markdown", - "id": "9b9ec0a5", + "id": "43987fb6", "metadata": { "editable": true }, @@ -4605,7 +4605,7 @@ }, { "cell_type": "markdown", - "id": "4aef81cc", + "id": "78741f5d", "metadata": { "editable": true }, @@ -4617,7 +4617,7 @@ }, { "cell_type": "markdown", - "id": "0e95f57e", + "id": "ecf46f39", "metadata": { "editable": true }, @@ -4627,7 +4627,7 @@ }, { "cell_type": "markdown", - "id": "270200f7", + "id": "f98d1ea2", "metadata": { "editable": true }, @@ -4640,7 +4640,7 @@ }, { "cell_type": "markdown", - "id": "cc8bdd04", + "id": "56a4e22a", "metadata": { "editable": true }, @@ -4650,7 +4650,7 @@ }, { "cell_type": "markdown", - "id": "e9de6d93", + "id": "3087fc2c", "metadata": { "editable": true }, @@ -4662,7 +4662,7 @@ }, { "cell_type": "markdown", - "id": "922b56ed", + "id": "8d386a35", "metadata": { "editable": true }, @@ -4675,7 +4675,7 @@ }, { "cell_type": "markdown", - "id": "b0692060", + "id": "c8d58182", "metadata": { "editable": true }, @@ -4688,7 +4688,7 @@ }, { "cell_type": "markdown", - "id": "b5f3bc64", + "id": "15fe7dd6", "metadata": { "editable": true }, @@ -4700,7 +4700,7 @@ }, { "cell_type": "markdown", - "id": "85737746", + "id": "042bbee6", "metadata": { "editable": true }, @@ -4712,7 +4712,7 @@ }, { "cell_type": "markdown", - "id": "8648f64b", + "id": "911f8e5c", "metadata": { "editable": true }, @@ -4722,7 +4722,7 @@ }, { "cell_type": "markdown", - "id": "f338de78", + "id": "40d4bbda", "metadata": { "editable": true }, @@ -4735,7 +4735,7 @@ }, { "cell_type": "markdown", - "id": "24410a0c", + "id": "8a00d027", "metadata": { "editable": true }, @@ -4747,7 +4747,7 @@ }, { "cell_type": "markdown", - "id": "62a8b683", + "id": "ff109317", "metadata": { "editable": true }, @@ -4759,7 +4759,7 @@ }, { "cell_type": "markdown", - "id": "6faeefbc", + "id": "4e5859eb", "metadata": { "editable": true }, @@ -4776,7 +4776,7 @@ }, { "cell_type": "markdown", - "id": "07b2588e", + "id": "75a8694b", "metadata": { "editable": true }, @@ -4788,7 +4788,7 @@ }, { "cell_type": "markdown", - "id": "37216e44", + "id": "80c47724", "metadata": { "editable": true }, @@ -4798,7 +4798,7 @@ }, { "cell_type": "markdown", - "id": "ef41b3eb", + "id": "6b385376", "metadata": { "editable": true }, @@ -4810,7 +4810,7 @@ }, { "cell_type": "markdown", - "id": "a01bade9", + "id": "32963935", "metadata": { "editable": true }, @@ -4820,7 +4820,7 @@ }, { "cell_type": "markdown", - "id": "d29ddf13", + "id": "de8c6756", "metadata": { "editable": true }, @@ -4832,7 +4832,7 @@ }, { "cell_type": "markdown", - "id": "91bdf67a", + "id": "76b587d1", "metadata": { "editable": true }, @@ -4844,7 +4844,7 @@ }, { "cell_type": "markdown", - "id": "5a44ca2f", + "id": "8a1fc10c", "metadata": { "editable": true }, @@ -4860,7 +4860,7 @@ }, { "cell_type": "markdown", - "id": "8f066cf6", + "id": "a6377518", "metadata": { "editable": true }, @@ -4872,7 +4872,7 @@ }, { "cell_type": "markdown", - "id": "7aee679c", + "id": "367d1837", "metadata": { "editable": true }, @@ -4884,7 +4884,7 @@ }, { "cell_type": "markdown", - "id": "ae1a53ff", + "id": "2fa779fe", "metadata": { "editable": true }, @@ -4894,7 +4894,7 @@ }, { "cell_type": "markdown", - "id": "3871adb7", + "id": "e0f99ac6", "metadata": { "editable": true }, @@ -4906,7 +4906,7 @@ }, { "cell_type": "markdown", - "id": "8ba7c52e", + "id": "d2134909", "metadata": { "editable": true }, @@ -4916,7 +4916,7 @@ }, { "cell_type": "markdown", - "id": "963d7c31", + "id": "09b0f22e", "metadata": { "editable": true }, @@ -4928,7 +4928,7 @@ }, { "cell_type": "markdown", - "id": "93d2010b", + "id": "1d1987fc", "metadata": { "editable": true }, @@ -4945,7 +4945,7 @@ }, { "cell_type": "markdown", - "id": "d18dae13", + "id": "16f94ea4", "metadata": { "editable": true }, @@ -4957,7 +4957,7 @@ }, { "cell_type": "markdown", - "id": "cfebf7d2", + "id": "21c224fd", "metadata": { "editable": true }, @@ -4969,7 +4969,7 @@ }, { "cell_type": "markdown", - "id": "53d5aa54", + "id": "80c9611e", "metadata": { "editable": true }, @@ -4979,7 +4979,7 @@ }, { "cell_type": "markdown", - "id": "68871be0", + "id": "ca8706ca", "metadata": { "editable": true }, @@ -4991,7 +4991,7 @@ }, { "cell_type": "markdown", - "id": "0f010d4c", + "id": "6bcce614", "metadata": { "editable": true }, @@ -5001,7 +5001,7 @@ }, { "cell_type": "markdown", - "id": "b0037b04", + "id": "857b9ab1", "metadata": { "editable": true }, @@ -5013,7 +5013,7 @@ }, { "cell_type": "markdown", - "id": "cc6232af", + "id": "bb8c1546", "metadata": { "editable": true }, @@ -5023,7 +5023,7 @@ }, { "cell_type": "markdown", - "id": "34859c6d", + "id": "2f7636ba", "metadata": { "editable": true }, @@ -5035,7 +5035,7 @@ }, { "cell_type": "markdown", - "id": "2ac234a9", + "id": "07e308e9", "metadata": { "editable": true }, @@ -5045,7 +5045,7 @@ }, { "cell_type": "markdown", - "id": "623c704e", + "id": "7a20da6b", "metadata": { "editable": true }, @@ -5057,13 +5057,856 @@ }, { "cell_type": "markdown", - "id": "05c9947e", + "id": "404a4814", "metadata": { "editable": true }, "source": [ "This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package [CVXOPT](https://cvxopt.org/). We will discuss how to code LASSO regression next week, when we have introduced gradient methods." ] + }, + { + "cell_type": "markdown", + "id": "c47881af", + "metadata": { + "editable": true + }, + "source": [ + "## Material for exercises week 35" + ] + }, + { + "cell_type": "markdown", + "id": "de625607", + "metadata": { + "editable": true + }, + "source": [ + "## Important technicalities: More on Rescaling data\n", + "\n", + "When you are comparing your own code with for example **Scikit-Learn**'s\n", + "library, there are some technicalities to keep in mind. The examples\n", + "here demonstrate some of these aspects with potential pitfalls.\n", + "\n", + "The discussion here focuses on the role of the intercept, how we can\n", + "set up the design matrix, what scaling we should use and other topics\n", + "which tend confuse us.\n", + "\n", + "The intercept can be interpreted as the expected value of our\n", + "target/output variables when all other predictors are set to zero.\n", + "Thus, if we cannot assume that the expected outputs/targets are zero\n", + "when all predictors are zero (the columns in the design matrix), it\n", + "may be a bad idea to implement a model which penalizes the intercept.\n", + "Furthermore, in for example Ridge and Lasso regression, the default solutions\n", + "from the library **Scikit-Learn** (when not shrinking $\\beta_0$) for the unknown parameters\n", + "$\\boldsymbol{\\beta}$, are derived under the assumption that both $\\boldsymbol{y}$ and\n", + "$\\boldsymbol{X}$ are zero centered, that is we subtract the mean values.\n", + "\n", + "If our predictors represent different scales, then it is important to\n", + "standardize the design matrix $\\boldsymbol{X}$ by subtracting the mean of each\n", + "column from the corresponding column and dividing the column with its\n", + "standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,\n", + "the results may differ. \n", + "\n", + "The\n", + "[Standardscaler](https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html)\n", + "function in **Scikit-Learn** does this for us. For the data sets we\n", + "have been studying in our various examples, the data are in many cases\n", + "already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a\n", + "survey of your data, with a critical assessment of them in case you need to scale the data.\n", + "\n", + "If you need to scale the data, not doing so will give an *unfair*\n", + "penalization of the parameters since their magnitude depends on the\n", + "scale of their corresponding predictor.\n", + "\n", + "The **Scikit-Learn** site has a good discussion of different ways of preprocessing data.\n", + "\n", + "Suppose as an example that you \n", + "you have an input variable given by the heights of different persons.\n", + "Human height might be measured in inches or meters or\n", + "kilometers. If measured in kilometers, a standard linear regression\n", + "model with this predictor would probably give a much bigger\n", + "coefficient term, than if measured in millimeters.\n", + "This can clearly lead to problems in evaluating the cost/loss functions.\n", + "\n", + "Keep in mind that when you transform your data set before training a model, the same transformation needs to be done\n", + "on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as" + ] + }, + { + "cell_type": "code", + "execution_count": 17, + "id": "c00a9805", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"\n", + "#Model training, we compute the mean value of y and X\n", + "y_train_mean = np.mean(y_train)\n", + "X_train_mean = np.mean(X_train,axis=0)\n", + "X_train = X_train - X_train_mean\n", + "y_train = y_train - y_train_mean\n", + "\n", + "# The we fit our model with the training data\n", + "trained_model = some_model.fit(X_train,y_train)\n", + "\n", + "\n", + "#Model prediction, we need also to transform our data set used for the prediction.\n", + "X_test = X_test - X_train_mean #Use mean from training data\n", + "y_pred = trained_model(X_test)\n", + "y_pred = y_pred + y_train_mean\n", + "\"\"\"" + ] + }, + { + "cell_type": "markdown", + "id": "b3fb9183", + "metadata": { + "editable": true + }, + "source": [ + "Let us try to understand what this may imply mathematically when we\n", + "subtract the mean values, also known as *zero centering*. For\n", + "simplicity, we will focus on ordinary regression, as done in the above example.\n", + "\n", + "The cost/loss function for regression is" + ] + }, + { + "cell_type": "markdown", + "id": "07a54955", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "C(\\beta_0, \\beta_1, ... , \\beta_{p-1}) = \\frac{1}{n}\\sum_{i=0}^{n} \\left(y_i - \\beta_0 - \\sum_{j=1}^{p-1} X_{ij}\\beta_j\\right)^2,.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1ac9c31d", + "metadata": { + "editable": true + }, + "source": [ + "Recall also that we use the squared value. This expression can lead to an\n", + "increased penalty for higher differences between predicted and\n", + "output/target values.\n", + "\n", + "What we have done is to single out the $\\beta_0$ term in the\n", + "definition of the mean squared error (MSE). The design matrix $X$\n", + "does in this case not contain any intercept column. When we take the\n", + "derivative with respect to $\\beta_0$, we want the derivative to obey" + ] + }, + { + "cell_type": "markdown", + "id": "ee6c6a4c", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial C}{\\partial \\beta_j} = 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "beb40c81", + "metadata": { + "editable": true + }, + "source": [ + "for all $j$. For $\\beta_0$ we have" + ] + }, + { + "cell_type": "markdown", + "id": "b051946e", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial C}{\\partial \\beta_0} = -\\frac{2}{n}\\sum_{i=0}^{n-1} \\left(y_i - \\beta_0 - \\sum_{j=1}^{p-1} X_{ij} \\beta_j\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "885a6b08", + "metadata": { + "editable": true + }, + "source": [ + "Multiplying away the constant $2/n$, we obtain" + ] + }, + { + "cell_type": "markdown", + "id": "5774d6be", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\sum_{i=0}^{n-1} \\beta_0 = \\sum_{i=0}^{n-1}y_i - \\sum_{i=0}^{n-1} \\sum_{j=1}^{p-1} X_{ij} \\beta_j.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "73ff6979", + "metadata": { + "editable": true + }, + "source": [ + "Let us specialize first to the case where we have only two parameters $\\beta_0$ and $\\beta_1$.\n", + "Our result for $\\beta_0$ simplifies then to" + ] + }, + { + "cell_type": "markdown", + "id": "792128bf", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "n\\beta_0 = \\sum_{i=0}^{n-1}y_i - \\sum_{i=0}^{n-1} X_{i1} \\beta_1.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "2be70ed5", + "metadata": { + "editable": true + }, + "source": [ + "We obtain then" + ] + }, + { + "cell_type": "markdown", + "id": "1d0c4ac1", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1}y_i - \\beta_1\\frac{1}{n}\\sum_{i=0}^{n-1} X_{i1}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "2d4a5936", + "metadata": { + "editable": true + }, + "source": [ + "If we define" + ] + }, + { + "cell_type": "markdown", + "id": "d30abc39", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\mu_{\\boldsymbol{x}_1}=\\frac{1}{n}\\sum_{i=0}^{n-1} X_{i1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "7f945f8b", + "metadata": { + "editable": true + }, + "source": [ + "and the mean value of the outputs as" + ] + }, + { + "cell_type": "markdown", + "id": "dff5bf57", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\mu_y=\\frac{1}{n}\\sum_{i=0}^{n-1}y_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "f222bcca", + "metadata": { + "editable": true + }, + "source": [ + "we have" + ] + }, + { + "cell_type": "markdown", + "id": "f1658ccd", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\mu_y - \\beta_1\\mu_{\\boldsymbol{x}_1}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "394ef3c4", + "metadata": { + "editable": true + }, + "source": [ + "In the general case with more parameters than $\\beta_0$ and $\\beta_1$, we have" + ] + }, + { + "cell_type": "markdown", + "id": "5f1dd1af", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1}y_i - \\frac{1}{n}\\sum_{i=0}^{n-1}\\sum_{j=1}^{p-1} X_{ij}\\beta_j.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0cafc2d3", + "metadata": { + "editable": true + }, + "source": [ + "We can rewrite the latter equation as" + ] + }, + { + "cell_type": "markdown", + "id": "2c878f98", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1}y_i - \\sum_{j=1}^{p-1} \\mu_{\\boldsymbol{x}_j}\\beta_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1dab41db", + "metadata": { + "editable": true + }, + "source": [ + "where we have defined" + ] + }, + { + "cell_type": "markdown", + "id": "cf2f2908", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\mu_{\\boldsymbol{x}_j}=\\frac{1}{n}\\sum_{i=0}^{n-1} X_{ij},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "20bac2a4", + "metadata": { + "editable": true + }, + "source": [ + "the mean value for all elements of the column vector $\\boldsymbol{x}_j$.\n", + "\n", + "Replacing $y_i$ with $y_i - y_i - \\overline{\\boldsymbol{y}}$ and centering also our design matrix results in a cost function (in vector-matrix disguise)" + ] + }, + { + "cell_type": "markdown", + "id": "0285afd3", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta}) = (\\boldsymbol{\\tilde{y}} - \\tilde{X}\\boldsymbol{\\beta})^T(\\boldsymbol{\\tilde{y}} - \\tilde{X}\\boldsymbol{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "ffff8823", + "metadata": { + "editable": true + }, + "source": [ + "If we minimize with respect to $\\boldsymbol{\\beta}$ we have then" + ] + }, + { + "cell_type": "markdown", + "id": "eea815a7", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\hat{\\boldsymbol{\\beta}} = (\\tilde{X}^T\\tilde{X})^{-1}\\tilde{X}^T\\boldsymbol{\\tilde{y}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0d935c60", + "metadata": { + "editable": true + }, + "source": [ + "where $\\boldsymbol{\\tilde{y}} = \\boldsymbol{y} - \\overline{\\boldsymbol{y}}$\n", + "and $\\tilde{X}_{ij} = X_{ij} - \\frac{1}{n}\\sum_{k=0}^{n-1}X_{kj}$.\n", + "\n", + "For Ridge regression we need to add $\\lambda \\boldsymbol{\\beta}^T\\boldsymbol{\\beta}$ to the cost function and get then" + ] + }, + { + "cell_type": "markdown", + "id": "bb4eabb8", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\hat{\\boldsymbol{\\beta}} = (\\tilde{X}^T\\tilde{X} + \\lambda I)^{-1}\\tilde{X}^T\\boldsymbol{\\tilde{y}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1da18d81", + "metadata": { + "editable": true + }, + "source": [ + "What does this mean? And why do we insist on all this? Let us look at some examples.\n", + "\n", + "This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*). Here our scaling of the data is done by subtracting the mean values only.\n", + "Note also that we do not split the data into training and test." + ] + }, + { + "cell_type": "code", + "execution_count": 18, + "id": "a11f0699", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "\n", + "np.random.seed(2021)\n", + "\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "\n", + "def fit_beta(X, y):\n", + " return np.linalg.pinv(X.T @ X) @ X.T @ y\n", + "\n", + "\n", + "true_beta = [2, 0.5, 3.7]\n", + "\n", + "x = np.linspace(0, 1, 11)\n", + "y = np.sum(\n", + " np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0\n", + ") + 0.1 * np.random.normal(size=len(x))\n", + "\n", + "degree = 3\n", + "X = np.zeros((len(x), degree))\n", + "\n", + "# Include the intercept in the design matrix\n", + "for p in range(degree):\n", + " X[:, p] = x ** p\n", + "\n", + "beta = fit_beta(X, y)\n", + "\n", + "# Intercept is included in the design matrix\n", + "skl = LinearRegression(fit_intercept=False).fit(X, y)\n", + "\n", + "print(f\"True beta: {true_beta}\")\n", + "print(f\"Fitted beta: {beta}\")\n", + "print(f\"Sklearn fitted beta: {skl.coef_}\")\n", + "ypredictOwn = X @ beta\n", + "ypredictSKL = skl.predict(X)\n", + "print(f\"MSE with intercept column\")\n", + "print(MSE(y,ypredictOwn))\n", + "print(f\"MSE with intercept column from SKL\")\n", + "print(MSE(y,ypredictSKL))\n", + "\n", + "\n", + "plt.figure()\n", + "plt.scatter(x, y, label=\"Data\")\n", + "plt.plot(x, X @ beta, label=\"Fit\")\n", + "plt.plot(x, skl.predict(X), label=\"Sklearn (fit_intercept=False)\")\n", + "\n", + "\n", + "# Do not include the intercept in the design matrix\n", + "X = np.zeros((len(x), degree - 1))\n", + "\n", + "for p in range(degree - 1):\n", + " X[:, p] = x ** (p + 1)\n", + "\n", + "# Intercept is not included in the design matrix\n", + "skl = LinearRegression(fit_intercept=True).fit(X, y)\n", + "\n", + "# Use centered values for X and y when computing coefficients\n", + "y_offset = np.average(y, axis=0)\n", + "X_offset = np.average(X, axis=0)\n", + "\n", + "beta = fit_beta(X - X_offset, y - y_offset)\n", + "intercept = np.mean(y_offset - X_offset @ beta)\n", + "\n", + "print(f\"Manual intercept: {intercept}\")\n", + "print(f\"Fitted beta (without intercept): {beta}\")\n", + "print(f\"Sklearn intercept: {skl.intercept_}\")\n", + "print(f\"Sklearn fitted beta (without intercept): {skl.coef_}\")\n", + "ypredictOwn = X @ beta\n", + "ypredictSKL = skl.predict(X)\n", + "print(f\"MSE with Manual intercept\")\n", + "print(MSE(y,ypredictOwn+intercept))\n", + "print(f\"MSE with Sklearn intercept\")\n", + "print(MSE(y,ypredictSKL))\n", + "\n", + "plt.plot(x, X @ beta + intercept, \"--\", label=\"Fit (manual intercept)\")\n", + "plt.plot(x, skl.predict(X), \"--\", label=\"Sklearn (fit_intercept=True)\")\n", + "plt.grid()\n", + "plt.legend()\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "ed61dd49", + "metadata": { + "editable": true + }, + "source": [ + "The intercept is the value of our output/target variable\n", + "when all our features are zero and our function crosses the $y$-axis (for a one-dimensional case). \n", + "\n", + "Printing the MSE, we see first that both methods give the same MSE, as\n", + "they should. However, when we move to for example Ridge regression,\n", + "the way we treat the intercept may give a larger or smaller MSE,\n", + "meaning that the MSE can be penalized by the value of the\n", + "intercept. Not including the intercept in the fit, means that the\n", + "regularization term does not include $\\beta_0$. For different values\n", + "of $\\lambda$, this may lead to different MSE values. \n", + "\n", + "To remind the reader, the regularization term, with the intercept in Ridge regression, is given by" + ] + }, + { + "cell_type": "markdown", + "id": "5de45190", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda \\vert\\vert \\boldsymbol{\\beta} \\vert\\vert_2^2 = \\lambda \\sum_{j=0}^{p-1}\\beta_j^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "6f53e8c1", + "metadata": { + "editable": true + }, + "source": [ + "but when we take out the intercept, this equation becomes" + ] + }, + { + "cell_type": "markdown", + "id": "3b0ffc74", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda \\vert\\vert \\boldsymbol{\\beta} \\vert\\vert_2^2 = \\lambda \\sum_{j=1}^{p-1}\\beta_j^2.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "10305919", + "metadata": { + "editable": true + }, + "source": [ + "For Lasso regression we have" + ] + }, + { + "cell_type": "markdown", + "id": "b2ce8814", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda \\vert\\vert \\boldsymbol{\\beta} \\vert\\vert_1 = \\lambda \\sum_{j=1}^{p-1}\\vert\\beta_j\\vert.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "e08c9fd7", + "metadata": { + "editable": true + }, + "source": [ + "It means that, when scaling the design matrix and the outputs/targets,\n", + "by subtracting the mean values, we have an optimization problem which\n", + "is not penalized by the intercept. The MSE value can then be smaller\n", + "since it focuses only on the remaining quantities. If we however bring\n", + "back the intercept, we will get a MSE which then contains the\n", + "intercept.\n", + "\n", + "Armed with this wisdom, we attempt first to simply set the intercept equal to **False** in our implementation of Ridge regression for our well-known vanilla data set." + ] + }, + { + "cell_type": "code", + "execution_count": 19, + "id": "5b7e1c63", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn import linear_model\n", + "\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "# Useful for eventual debugging.\n", + "np.random.seed(3155)\n", + "\n", + "n = 100\n", + "x = np.random.rand(n)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n", + "\n", + "Maxpolydegree = 20\n", + "X = np.zeros((n,Maxpolydegree))\n", + "#We include explicitely the intercept column\n", + "for degree in range(Maxpolydegree):\n", + " X[:,degree] = x**degree\n", + "# We split the data in test and training data\n", + "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n", + "\n", + "p = Maxpolydegree\n", + "I = np.eye(p,p)\n", + "# Decide which values of lambda to use\n", + "nlambdas = 6\n", + "MSEOwnRidgePredict = np.zeros(nlambdas)\n", + "MSERidgePredict = np.zeros(nlambdas)\n", + "lambdas = np.logspace(-4, 2, nlambdas)\n", + "for i in range(nlambdas):\n", + " lmb = lambdas[i]\n", + " OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train\n", + " # Note: we include the intercept column and no scaling\n", + " RegRidge = linear_model.Ridge(lmb,fit_intercept=False)\n", + " RegRidge.fit(X_train,y_train)\n", + " # and then make the prediction\n", + " ytildeOwnRidge = X_train @ OwnRidgeBeta\n", + " ypredictOwnRidge = X_test @ OwnRidgeBeta\n", + " ytildeRidge = RegRidge.predict(X_train)\n", + " ypredictRidge = RegRidge.predict(X_test)\n", + " MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n", + " MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n", + " print(\"Beta values for own Ridge implementation\")\n", + " print(OwnRidgeBeta)\n", + " print(\"Beta values for Scikit-Learn Ridge implementation\")\n", + " print(RegRidge.coef_)\n", + " print(\"MSE values for own Ridge implementation\")\n", + " print(MSEOwnRidgePredict[i])\n", + " print(\"MSE values for Scikit-Learn Ridge implementation\")\n", + " print(MSERidgePredict[i])\n", + "\n", + "# Now plot the results\n", + "plt.figure()\n", + "plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')\n", + "plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')\n", + "\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('MSE')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "bbf9639e", + "metadata": { + "editable": true + }, + "source": [ + "The results here agree when we force **Scikit-Learn**'s Ridge function to include the first column in our design matrix.\n", + "We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.\n", + "What happens if we do not include the intercept in our fit?\n", + "Let us see how we can change this code by zero centering." + ] + }, + { + "cell_type": "code", + "execution_count": 20, + "id": "3a82ee91", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn import linear_model\n", + "from sklearn.preprocessing import StandardScaler\n", + "\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "# Useful for eventual debugging.\n", + "np.random.seed(315)\n", + "\n", + "n = 100\n", + "x = np.random.rand(n)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n", + "\n", + "Maxpolydegree = 20\n", + "X = np.zeros((n,Maxpolydegree-1))\n", + "\n", + "for degree in range(1,Maxpolydegree): #No intercept column\n", + " X[:,degree-1] = x**(degree)\n", + "\n", + "# We split the data in test and training data\n", + "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n", + "\n", + "#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable\n", + "X_train_mean = np.mean(X_train,axis=0)\n", + "#Center by removing mean from each feature\n", + "X_train_scaled = X_train - X_train_mean \n", + "X_test_scaled = X_test - X_train_mean\n", + "#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)\n", + "#Remove the intercept from the training data.\n", + "y_scaler = np.mean(y_train) \n", + "y_train_scaled = y_train - y_scaler \n", + "\n", + "p = Maxpolydegree-1\n", + "I = np.eye(p,p)\n", + "# Decide which values of lambda to use\n", + "nlambdas = 6\n", + "MSEOwnRidgePredict = np.zeros(nlambdas)\n", + "MSERidgePredict = np.zeros(nlambdas)\n", + "\n", + "lambdas = np.logspace(-4, 2, nlambdas)\n", + "for i in range(nlambdas):\n", + " lmb = lambdas[i]\n", + " OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)\n", + " intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data\n", + " #Add intercept to prediction\n", + " ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler \n", + " RegRidge = linear_model.Ridge(lmb)\n", + " RegRidge.fit(X_train,y_train)\n", + " ypredictRidge = RegRidge.predict(X_test)\n", + " MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n", + " MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n", + " print(\"Beta values for own Ridge implementation\")\n", + " print(OwnRidgeBeta) #Intercept is given by mean of target variable\n", + " print(\"Beta values for Scikit-Learn Ridge implementation\")\n", + " print(RegRidge.coef_)\n", + " print('Intercept from own implementation:')\n", + " print(intercept_)\n", + " print('Intercept from Scikit-Learn Ridge implementation')\n", + " print(RegRidge.intercept_)\n", + " print(\"MSE values for own Ridge implementation\")\n", + " print(MSEOwnRidgePredict[i])\n", + " print(\"MSE values for Scikit-Learn Ridge implementation\")\n", + " print(MSERidgePredict[i])\n", + "\n", + "\n", + "# Now plot the results\n", + "plt.figure()\n", + "plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')\n", + "plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('MSE')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "c061da66", + "metadata": { + "editable": true + }, + "source": [ + "We see here, when compared to the code which includes explicitely the\n", + "intercept column, that our MSE value is actually smaller. This is\n", + "because the regularization term does not include the intercept value\n", + "$\\beta_0$ in the fitting. This applies to Lasso regularization as\n", + "well. It means that our optimization is now done only with the\n", + "centered matrix and/or vector that enter the fitting procedure." + ] } ], "metadata": {}, diff --git a/doc/src/week35/week35.do.txt b/doc/src/week35/week35.do.txt index 9f66a608a..4309600e7 100644 --- a/doc/src/week35/week35.do.txt +++ b/doc/src/week35/week35.do.txt @@ -2320,3 +2320,505 @@ We can redefine $\lambda$ to absorb the constant $n/2$ and we rewrite the last e This equation does not lead to a nice analytical equation as in either Ridge regression or ordinary least squares. This equation can however be solved by using standard convex optimization algorithms using for example the Python package "CVXOPT":"https://cvxopt.org/". We will discuss how to code LASSO regression next week, when we have introduced gradient methods. + +!split +===== Material for exercises week 35 ===== + + +===== Important technicalities: More on Rescaling data ===== + + +When you are comparing your own code with for example _Scikit-Learn_'s +library, there are some technicalities to keep in mind. The examples +here demonstrate some of these aspects with potential pitfalls. + +The discussion here focuses on the role of the intercept, how we can +set up the design matrix, what scaling we should use and other topics +which tend confuse us. + +The intercept can be interpreted as the expected value of our +target/output variables when all other predictors are set to zero. +Thus, if we cannot assume that the expected outputs/targets are zero +when all predictors are zero (the columns in the design matrix), it +may be a bad idea to implement a model which penalizes the intercept. +Furthermore, in for example Ridge and Lasso regression, the default solutions +from the library _Scikit-Learn_ (when not shrinking $\beta_0$) for the unknown parameters +$\bm{\beta}$, are derived under the assumption that both $\bm{y}$ and +$\bm{X}$ are zero centered, that is we subtract the mean values. + + +If our predictors represent different scales, then it is important to +standardize the design matrix $\bm{X}$ by subtracting the mean of each +column from the corresponding column and dividing the column with its +standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library, +the results may differ. + +The +"Standardscaler":"https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" +function in _Scikit-Learn_ does this for us. For the data sets we +have been studying in our various examples, the data are in many cases +already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a +survey of your data, with a critical assessment of them in case you need to scale the data. + +If you need to scale the data, not doing so will give an *unfair* +penalization of the parameters since their magnitude depends on the +scale of their corresponding predictor. + +The _Scikit-Learn_ site URL:"https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section" has a good discussion of different ways of preprocessing data. + +Suppose as an example that you +you have an input variable given by the heights of different persons. +Human height might be measured in inches or meters or +kilometers. If measured in kilometers, a standard linear regression +model with this predictor would probably give a much bigger +coefficient term, than if measured in millimeters. +This can clearly lead to problems in evaluating the cost/loss functions. + + + +Keep in mind that when you transform your data set before training a model, the same transformation needs to be done +on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as + +!bc pycod +""" +#Model training, we compute the mean value of y and X +y_train_mean = np.mean(y_train) +X_train_mean = np.mean(X_train,axis=0) +X_train = X_train - X_train_mean +y_train = y_train - y_train_mean + +# The we fit our model with the training data +trained_model = some_model.fit(X_train,y_train) + + +#Model prediction, we need also to transform our data set used for the prediction. +X_test = X_test - X_train_mean #Use mean from training data +y_pred = trained_model(X_test) +y_pred = y_pred + y_train_mean +""" +!ec + + + +Let us try to understand what this may imply mathematically when we +subtract the mean values, also known as *zero centering*. For +simplicity, we will focus on ordinary regression, as done in the above example. + +The cost/loss function for regression is +!bt +\[ +C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,. +\] +!et + +Recall also that we use the squared value. This expression can lead to an +increased penalty for higher differences between predicted and +output/target values. + +What we have done is to single out the $\beta_0$ term in the +definition of the mean squared error (MSE). The design matrix $X$ +does in this case not contain any intercept column. When we take the +derivative with respect to $\beta_0$, we want the derivative to obey + +!bt +\[ +\frac{\partial C}{\partial \beta_j} = 0, +\] +!et + +for all $j$. For $\beta_0$ we have + +!bt +\[ +\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right). +\] +!et +Multiplying away the constant $2/n$, we obtain +!bt +\[ +\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j. +\] +!et + +Let us specialize first to the case where we have only two parameters $\beta_0$ and $\beta_1$. +Our result for $\beta_0$ simplifies then to +!bt +\[ +n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1. +\] +!et +We obtain then +!bt +\[ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}. +\] +!et +If we define +!bt +\[ +\mu_{\bm{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}, +\] +!et +and the mean value of the outputs as +!bt +\[ +\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i, +\] +!et +we have +!bt +\[ +\beta_0 = \mu_y - \beta_1\mu_{\bm{x}_1}. +\] +!et +In the general case with more parameters than $\beta_0$ and $\beta_1$, we have +!bt +\[ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j. +\] +!et + +We can rewrite the latter equation as +!bt +\[ +\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\bm{x}_j}\beta_j, +\] +!et +where we have defined +!bt +\[ +\mu_{\bm{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij}, +\] +!et +the mean value for all elements of the column vector $\bm{x}_j$. + + + +Replacing $y_i$ with $y_i - y_i - \overline{\bm{y}}$ and centering also our design matrix results in a cost function (in vector-matrix disguise) +!bt +\[ +C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}). +\] +!et + + + +If we minimize with respect to $\bm{\beta}$ we have then + +!bt +\[ +\hat{\bm{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}, +\] +!et + +where $\boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\bm{y}}$ +and $\tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj}$. + +For Ridge regression we need to add $\lambda \boldsymbol{\beta}^T\boldsymbol{\beta}$ to the cost function and get then +!bt +\[ +\hat{\bm{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}. +\] +!et + +What does this mean? And why do we insist on all this? Let us look at some examples. + + +This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*). Here our scaling of the data is done by subtracting the mean values only. +Note also that we do not split the data into training and test. + +!bc pycod +import numpy as np +import matplotlib.pyplot as plt + +from sklearn.linear_model import LinearRegression + + +np.random.seed(2021) + +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + + +def fit_beta(X, y): + return np.linalg.pinv(X.T @ X) @ X.T @ y + + +true_beta = [2, 0.5, 3.7] + +x = np.linspace(0, 1, 11) +y = np.sum( + np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0 +) + 0.1 * np.random.normal(size=len(x)) + +degree = 3 +X = np.zeros((len(x), degree)) + +# Include the intercept in the design matrix +for p in range(degree): + X[:, p] = x ** p + +beta = fit_beta(X, y) + +# Intercept is included in the design matrix +skl = LinearRegression(fit_intercept=False).fit(X, y) + +print(f"True beta: {true_beta}") +print(f"Fitted beta: {beta}") +print(f"Sklearn fitted beta: {skl.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with intercept column") +print(MSE(y,ypredictOwn)) +print(f"MSE with intercept column from SKL") +print(MSE(y,ypredictSKL)) + + +plt.figure() +plt.scatter(x, y, label="Data") +plt.plot(x, X @ beta, label="Fit") +plt.plot(x, skl.predict(X), label="Sklearn (fit_intercept=False)") + + +# Do not include the intercept in the design matrix +X = np.zeros((len(x), degree - 1)) + +for p in range(degree - 1): + X[:, p] = x ** (p + 1) + +# Intercept is not included in the design matrix +skl = LinearRegression(fit_intercept=True).fit(X, y) + +# Use centered values for X and y when computing coefficients +y_offset = np.average(y, axis=0) +X_offset = np.average(X, axis=0) + +beta = fit_beta(X - X_offset, y - y_offset) +intercept = np.mean(y_offset - X_offset @ beta) + +print(f"Manual intercept: {intercept}") +print(f"Fitted beta (without intercept): {beta}") +print(f"Sklearn intercept: {skl.intercept_}") +print(f"Sklearn fitted beta (without intercept): {skl.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with Manual intercept") +print(MSE(y,ypredictOwn+intercept)) +print(f"MSE with Sklearn intercept") +print(MSE(y,ypredictSKL)) + +plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)") +plt.plot(x, skl.predict(X), "--", label="Sklearn (fit_intercept=True)") +plt.grid() +plt.legend() + +plt.show() + +!ec + +The intercept is the value of our output/target variable +when all our features are zero and our function crosses the $y$-axis (for a one-dimensional case). + +Printing the MSE, we see first that both methods give the same MSE, as +they should. However, when we move to for example Ridge regression, +the way we treat the intercept may give a larger or smaller MSE, +meaning that the MSE can be penalized by the value of the +intercept. Not including the intercept in the fit, means that the +regularization term does not include $\beta_0$. For different values +of $\lambda$, this may lead to different MSE values. + +To remind the reader, the regularization term, with the intercept in Ridge regression, is given by +!bt +\[ +\lambda \vert\vert \bm{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2, +\] +!et +but when we take out the intercept, this equation becomes +!bt +\[ +\lambda \vert\vert \bm{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2. +\] +!et + +For Lasso regression we have +!bt +\[ +\lambda \vert\vert \bm{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert. +\] +!et + +It means that, when scaling the design matrix and the outputs/targets, +by subtracting the mean values, we have an optimization problem which +is not penalized by the intercept. The MSE value can then be smaller +since it focuses only on the remaining quantities. If we however bring +back the intercept, we will get a MSE which then contains the +intercept. + + +Armed with this wisdom, we attempt first to simply set the intercept equal to _False_ in our implementation of Ridge regression for our well-known vanilla data set. + +!bc pycod +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.model_selection import train_test_split +from sklearn import linear_model + +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + + +# A seed just to ensure that the random numbers are the same for every run. +# Useful for eventual debugging. +np.random.seed(3155) + +n = 100 +x = np.random.rand(n) +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) + +Maxpolydegree = 20 +X = np.zeros((n,Maxpolydegree)) +#We include explicitely the intercept column +for degree in range(Maxpolydegree): + X[:,degree] = x**degree +# We split the data in test and training data +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2) + +p = Maxpolydegree +I = np.eye(p,p) +# Decide which values of lambda to use +nlambdas = 6 +MSEOwnRidgePredict = np.zeros(nlambdas) +MSERidgePredict = np.zeros(nlambdas) +lambdas = np.logspace(-4, 2, nlambdas) +for i in range(nlambdas): + lmb = lambdas[i] + OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train + # Note: we include the intercept column and no scaling + RegRidge = linear_model.Ridge(lmb,fit_intercept=False) + RegRidge.fit(X_train,y_train) + # and then make the prediction + ytildeOwnRidge = X_train @ OwnRidgeBeta + ypredictOwnRidge = X_test @ OwnRidgeBeta + ytildeRidge = RegRidge.predict(X_train) + ypredictRidge = RegRidge.predict(X_test) + MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge) + MSERidgePredict[i] = MSE(y_test,ypredictRidge) + print("Beta values for own Ridge implementation") + print(OwnRidgeBeta) + print("Beta values for Scikit-Learn Ridge implementation") + print(RegRidge.coef_) + print("MSE values for own Ridge implementation") + print(MSEOwnRidgePredict[i]) + print("MSE values for Scikit-Learn Ridge implementation") + print(MSERidgePredict[i]) + +# Now plot the results +plt.figure() +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test') +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test') + +plt.xlabel('log10(lambda)') +plt.ylabel('MSE') +plt.legend() +plt.show() + +!ec + +The results here agree when we force _Scikit-Learn_'s Ridge function to include the first column in our design matrix. +We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix. +What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering. + +!bc pycod +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.model_selection import train_test_split +from sklearn import linear_model +from sklearn.preprocessing import StandardScaler + +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n +# A seed just to ensure that the random numbers are the same for every run. +# Useful for eventual debugging. +np.random.seed(315) + +n = 100 +x = np.random.rand(n) +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) + +Maxpolydegree = 20 +X = np.zeros((n,Maxpolydegree-1)) + +for degree in range(1,Maxpolydegree): #No intercept column + X[:,degree-1] = x**(degree) + +# We split the data in test and training data +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2) + +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable +X_train_mean = np.mean(X_train,axis=0) +#Center by removing mean from each feature +X_train_scaled = X_train - X_train_mean +X_test_scaled = X_test - X_train_mean +#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered) +#Remove the intercept from the training data. +y_scaler = np.mean(y_train) +y_train_scaled = y_train - y_scaler + +p = Maxpolydegree-1 +I = np.eye(p,p) +# Decide which values of lambda to use +nlambdas = 6 +MSEOwnRidgePredict = np.zeros(nlambdas) +MSERidgePredict = np.zeros(nlambdas) + +lambdas = np.logspace(-4, 2, nlambdas) +for i in range(nlambdas): + lmb = lambdas[i] + OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled) + intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data + #Add intercept to prediction + ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler + RegRidge = linear_model.Ridge(lmb) + RegRidge.fit(X_train,y_train) + ypredictRidge = RegRidge.predict(X_test) + MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge) + MSERidgePredict[i] = MSE(y_test,ypredictRidge) + print("Beta values for own Ridge implementation") + print(OwnRidgeBeta) #Intercept is given by mean of target variable + print("Beta values for Scikit-Learn Ridge implementation") + print(RegRidge.coef_) + print('Intercept from own implementation:') + print(intercept_) + print('Intercept from Scikit-Learn Ridge implementation') + print(RegRidge.intercept_) + print("MSE values for own Ridge implementation") + print(MSEOwnRidgePredict[i]) + print("MSE values for Scikit-Learn Ridge implementation") + print(MSERidgePredict[i]) + + +# Now plot the results +plt.figure() +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test') +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test') +plt.xlabel('log10(lambda)') +plt.ylabel('MSE') +plt.legend() +plt.show() +!ec +We see here, when compared to the code which includes explicitely the +intercept column, that our MSE value is actually smaller. This is +because the regularization term does not include the intercept value +$\beta_0$ in the fitting. This applies to Lasso regularization as +well. It means that our optimization is now done only with the +centered matrix and/or vector that enter the fitting procedure. + + + +