diff --git a/doc/LectureNotes/DataFiles/cancer.dot b/doc/LectureNotes/DataFiles/cancer.dot index 75fc2296d..7174d42eb 100644 --- a/doc/LectureNotes/DataFiles/cancer.dot +++ b/doc/LectureNotes/DataFiles/cancer.dot @@ -10,7 +10,7 @@ edge [fontname="helvetica"] ; 2 -> 3 ; 4 [label="gini = 0.0\nsamples = 239\nvalue = [[239, 0]\n[0, 239]]", fillcolor="#e58139"] ; 3 -> 4 ; -5 [label="mean texture <= 18.935\ngini = 0.444\nsamples = 3\nvalue = [[2, 1]\n[1, 2]]", fillcolor="#fdf6f0"] ; +5 [label="mean radius <= 12.265\ngini = 0.444\nsamples = 3\nvalue = [[2, 1]\n[1, 2]]", fillcolor="#fdf6f0"] ; 3 -> 5 ; 6 [label="gini = 0.0\nsamples = 1\nvalue = [[0, 1]\n[1, 0]]", fillcolor="#e58139"] ; 5 -> 6 ; @@ -22,7 +22,7 @@ edge [fontname="helvetica"] ; 8 -> 9 ; 10 [label="gini = 0.0\nsamples = 3\nvalue = [[0, 3]\n[3, 0]]", fillcolor="#e58139"] ; 8 -> 10 ; -11 [label="mean texture <= 16.22\ngini = 0.278\nsamples = 6\nvalue = [[1, 5]\n[5, 1]]", fillcolor="#f4caac"] ; +11 [label="area error <= 13.475\ngini = 0.278\nsamples = 6\nvalue = [[1, 5]\n[5, 1]]", fillcolor="#f4caac"] ; 1 -> 11 ; 12 [label="gini = 0.0\nsamples = 1\nvalue = [[1, 0]\n[0, 1]]", fillcolor="#e58139"] ; 11 -> 12 ; @@ -30,11 +30,11 @@ edge [fontname="helvetica"] ; 11 -> 13 ; 14 [label="worst texture <= 20.645\ngini = 0.202\nsamples = 167\nvalue = [[19, 148]\n[148, 19]]", fillcolor="#f0b68c"] ; 0 -> 14 [labeldistance=2.5, labelangle=-45, headlabel="False"] ; -15 [label="worst radius <= 17.74\ngini = 0.375\nsamples = 16\nvalue = [[12, 4]\n[4, 12]]", fillcolor="#f9e3d4"] ; +15 [label="worst area <= 964.4\ngini = 0.375\nsamples = 16\nvalue = [[12, 4]\n[4, 12]]", fillcolor="#f9e3d4"] ; 14 -> 15 ; 16 [label="gini = 0.0\nsamples = 11\nvalue = [[11, 0]\n[0, 11]]", fillcolor="#e58139"] ; 15 -> 16 ; -17 [label="worst concave points <= 0.104\ngini = 0.32\nsamples = 5\nvalue = [[1, 4]\n[4, 1]]", fillcolor="#f6d5bd"] ; +17 [label="worst texture <= 18.445\ngini = 0.32\nsamples = 5\nvalue = [[1, 4]\n[4, 1]]", fillcolor="#f6d5bd"] ; 15 -> 17 ; 18 [label="gini = 0.0\nsamples = 1\nvalue = [[1, 0]\n[0, 1]]", fillcolor="#e58139"] ; 17 -> 18 ; @@ -42,7 +42,7 @@ edge [fontname="helvetica"] ; 17 -> 19 ; 20 [label="mean concave points <= 0.049\ngini = 0.088\nsamples = 151\nvalue = [[7, 144]\n[144, 7]]", fillcolor="#ea985d"] ; 14 -> 20 ; -21 [label="compactness error <= 0.016\ngini = 0.48\nsamples = 15\nvalue = [[6, 9]\n[9, 6]]", fillcolor="#ffffff"] ; +21 [label="concave points error <= 0.01\ngini = 0.48\nsamples = 15\nvalue = [[6, 9]\n[9, 6]]", fillcolor="#ffffff"] ; 20 -> 21 ; 22 [label="gini = 0.0\nsamples = 9\nvalue = [[0, 9]\n[9, 0]]", fillcolor="#e58139"] ; 21 -> 22 ; diff --git a/doc/LectureNotes/DataFiles/cancer.png b/doc/LectureNotes/DataFiles/cancer.png index 00a3090d5..4de674f5f 100644 Binary files a/doc/LectureNotes/DataFiles/cancer.png and b/doc/LectureNotes/DataFiles/cancer.png differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter1.doctree b/doc/LectureNotes/_build/.doctrees/chapter1.doctree index cd1ae20b2..3ed3101d9 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter1.doctree and b/doc/LectureNotes/_build/.doctrees/chapter1.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter10.doctree b/doc/LectureNotes/_build/.doctrees/chapter10.doctree index 0c5e9dd9c..efe414fa7 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter10.doctree and b/doc/LectureNotes/_build/.doctrees/chapter10.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter11.doctree b/doc/LectureNotes/_build/.doctrees/chapter11.doctree index 81592a76c..af6872c92 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter11.doctree and b/doc/LectureNotes/_build/.doctrees/chapter11.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter12.doctree b/doc/LectureNotes/_build/.doctrees/chapter12.doctree index 01a8d728b..2c41ee728 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter12.doctree and b/doc/LectureNotes/_build/.doctrees/chapter12.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter13.doctree b/doc/LectureNotes/_build/.doctrees/chapter13.doctree index b9e13342c..d66ea8801 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter13.doctree and b/doc/LectureNotes/_build/.doctrees/chapter13.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter2.doctree b/doc/LectureNotes/_build/.doctrees/chapter2.doctree index 4a4a2d964..310737fda 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter2.doctree and b/doc/LectureNotes/_build/.doctrees/chapter2.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter3.doctree b/doc/LectureNotes/_build/.doctrees/chapter3.doctree index 146a684ec..8b32ed0c3 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter3.doctree and b/doc/LectureNotes/_build/.doctrees/chapter3.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter6.doctree b/doc/LectureNotes/_build/.doctrees/chapter6.doctree index 8eefc7377..93107583d 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter6.doctree and b/doc/LectureNotes/_build/.doctrees/chapter6.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/chapter8.doctree b/doc/LectureNotes/_build/.doctrees/chapter8.doctree index d72bb0912..b2bc92020 100644 Binary files a/doc/LectureNotes/_build/.doctrees/chapter8.doctree and b/doc/LectureNotes/_build/.doctrees/chapter8.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/clustering.doctree b/doc/LectureNotes/_build/.doctrees/clustering.doctree index 8c06fadea..007fe246d 100644 Binary files a/doc/LectureNotes/_build/.doctrees/clustering.doctree and b/doc/LectureNotes/_build/.doctrees/clustering.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/environment.pickle b/doc/LectureNotes/_build/.doctrees/environment.pickle index f82fb6d9d..9107cd605 100644 Binary files a/doc/LectureNotes/_build/.doctrees/environment.pickle and b/doc/LectureNotes/_build/.doctrees/environment.pickle differ diff --git a/doc/LectureNotes/_build/.doctrees/exercisesweek41.doctree b/doc/LectureNotes/_build/.doctrees/exercisesweek41.doctree index cd456f04b..29fdc772c 100644 Binary files a/doc/LectureNotes/_build/.doctrees/exercisesweek41.doctree and b/doc/LectureNotes/_build/.doctrees/exercisesweek41.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/exercisesweek42.doctree b/doc/LectureNotes/_build/.doctrees/exercisesweek42.doctree index 22699b3eb..f0399fe60 100644 Binary files a/doc/LectureNotes/_build/.doctrees/exercisesweek42.doctree and b/doc/LectureNotes/_build/.doctrees/exercisesweek42.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/linalg.doctree b/doc/LectureNotes/_build/.doctrees/linalg.doctree index cc0d00311..9b3f5bc13 100644 Binary files a/doc/LectureNotes/_build/.doctrees/linalg.doctree and b/doc/LectureNotes/_build/.doctrees/linalg.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/statistics.doctree b/doc/LectureNotes/_build/.doctrees/statistics.doctree index b2db3d43c..f849f3f22 100644 Binary files a/doc/LectureNotes/_build/.doctrees/statistics.doctree and b/doc/LectureNotes/_build/.doctrees/statistics.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/week34.doctree b/doc/LectureNotes/_build/.doctrees/week34.doctree index 4a140e724..bd25631c0 100644 Binary files a/doc/LectureNotes/_build/.doctrees/week34.doctree and b/doc/LectureNotes/_build/.doctrees/week34.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/week35.doctree b/doc/LectureNotes/_build/.doctrees/week35.doctree index 18aaf5e2b..91b645df8 100644 Binary files a/doc/LectureNotes/_build/.doctrees/week35.doctree and b/doc/LectureNotes/_build/.doctrees/week35.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/week37.doctree b/doc/LectureNotes/_build/.doctrees/week37.doctree index 681f16e44..d887c8b6b 100644 Binary files a/doc/LectureNotes/_build/.doctrees/week37.doctree and b/doc/LectureNotes/_build/.doctrees/week37.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/week39.doctree b/doc/LectureNotes/_build/.doctrees/week39.doctree index f88d4eaed..dc560179b 100644 Binary files a/doc/LectureNotes/_build/.doctrees/week39.doctree and b/doc/LectureNotes/_build/.doctrees/week39.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/week40.doctree b/doc/LectureNotes/_build/.doctrees/week40.doctree index 70da268cc..33c1cf5ab 100644 Binary files a/doc/LectureNotes/_build/.doctrees/week40.doctree and b/doc/LectureNotes/_build/.doctrees/week40.doctree differ diff --git a/doc/LectureNotes/_build/.doctrees/week42.doctree b/doc/LectureNotes/_build/.doctrees/week42.doctree index f2eb64001..017d00a6e 100644 Binary files a/doc/LectureNotes/_build/.doctrees/week42.doctree and b/doc/LectureNotes/_build/.doctrees/week42.doctree differ diff --git a/doc/LectureNotes/_build/html/_images/chapter1_17_0.png b/doc/LectureNotes/_build/html/_images/chapter1_17_0.png index a29de85c7..65904afe4 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter1_17_0.png and b/doc/LectureNotes/_build/html/_images/chapter1_17_0.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter1_19_1.png b/doc/LectureNotes/_build/html/_images/chapter1_19_1.png index eec9b42d1..37061e76f 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter1_19_1.png and b/doc/LectureNotes/_build/html/_images/chapter1_19_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter1_33_0.png b/doc/LectureNotes/_build/html/_images/chapter1_33_0.png index b76788ee8..e053ae719 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter1_33_0.png and b/doc/LectureNotes/_build/html/_images/chapter1_33_0.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter1_9_0.png b/doc/LectureNotes/_build/html/_images/chapter1_9_0.png index e6ffd825c..1553c1bce 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter1_9_0.png and b/doc/LectureNotes/_build/html/_images/chapter1_9_0.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter2_252_2.png b/doc/LectureNotes/_build/html/_images/chapter2_252_2.png new file mode 100644 index 000000000..de1fd6ae7 Binary files /dev/null and b/doc/LectureNotes/_build/html/_images/chapter2_252_2.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter3_51_0.png b/doc/LectureNotes/_build/html/_images/chapter3_51_0.png index e7b53c753..ae6e2dab3 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter3_51_0.png and b/doc/LectureNotes/_build/html/_images/chapter3_51_0.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter3_66_6.png b/doc/LectureNotes/_build/html/_images/chapter3_66_6.png new file mode 100644 index 000000000..9ebeeb751 Binary files /dev/null and b/doc/LectureNotes/_build/html/_images/chapter3_66_6.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter6_1_1.png b/doc/LectureNotes/_build/html/_images/chapter6_1_1.png index 2b27c01d5..4154c917d 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter6_1_1.png and b/doc/LectureNotes/_build/html/_images/chapter6_1_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter6_1_2.png b/doc/LectureNotes/_build/html/_images/chapter6_1_2.png index de610b8d5..d3ad5e54e 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter6_1_2.png and b/doc/LectureNotes/_build/html/_images/chapter6_1_2.png differ diff --git a/doc/LectureNotes/_build/html/_images/chapter8_65_1.png b/doc/LectureNotes/_build/html/_images/chapter8_65_1.png index 96f175412..ec7c0c4d0 100644 Binary files a/doc/LectureNotes/_build/html/_images/chapter8_65_1.png and b/doc/LectureNotes/_build/html/_images/chapter8_65_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/exercisesweek41_16_2.png b/doc/LectureNotes/_build/html/_images/exercisesweek41_16_2.png index 0d1102c6b..6f7fb7a86 100644 Binary files a/doc/LectureNotes/_build/html/_images/exercisesweek41_16_2.png and b/doc/LectureNotes/_build/html/_images/exercisesweek41_16_2.png differ diff --git a/doc/LectureNotes/_build/html/_images/exercisesweek41_22_2.png b/doc/LectureNotes/_build/html/_images/exercisesweek41_22_2.png index 5840ca3f8..c8a22fb4b 100644 Binary files a/doc/LectureNotes/_build/html/_images/exercisesweek41_22_2.png and b/doc/LectureNotes/_build/html/_images/exercisesweek41_22_2.png differ diff --git a/doc/LectureNotes/_build/html/_images/exercisesweek41_5_1.png b/doc/LectureNotes/_build/html/_images/exercisesweek41_5_1.png index bc123afa5..8579745d2 100644 Binary files a/doc/LectureNotes/_build/html/_images/exercisesweek41_5_1.png and b/doc/LectureNotes/_build/html/_images/exercisesweek41_5_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/statistics_181_0.png b/doc/LectureNotes/_build/html/_images/statistics_181_0.png index a96d15c8d..7771c0877 100644 Binary files a/doc/LectureNotes/_build/html/_images/statistics_181_0.png and b/doc/LectureNotes/_build/html/_images/statistics_181_0.png differ diff --git a/doc/LectureNotes/_build/html/_images/statistics_188_1.png b/doc/LectureNotes/_build/html/_images/statistics_188_1.png index cb1a42624..9c63acba6 100644 Binary files a/doc/LectureNotes/_build/html/_images/statistics_188_1.png and b/doc/LectureNotes/_build/html/_images/statistics_188_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/week37_121_0.png b/doc/LectureNotes/_build/html/_images/week37_121_0.png index 59ff71db5..ecc705d16 100644 Binary files a/doc/LectureNotes/_build/html/_images/week37_121_0.png and b/doc/LectureNotes/_build/html/_images/week37_121_0.png differ diff --git a/doc/LectureNotes/_build/html/_images/week39_153_1.png b/doc/LectureNotes/_build/html/_images/week39_153_1.png index 717dbfc68..6be50b53d 100644 Binary files a/doc/LectureNotes/_build/html/_images/week39_153_1.png and b/doc/LectureNotes/_build/html/_images/week39_153_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/week39_166_1.png b/doc/LectureNotes/_build/html/_images/week39_166_1.png index b00dcb4a9..447c56361 100644 Binary files a/doc/LectureNotes/_build/html/_images/week39_166_1.png and b/doc/LectureNotes/_build/html/_images/week39_166_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/week40_100_1.png b/doc/LectureNotes/_build/html/_images/week40_100_1.png index 31858288a..f6741fa45 100644 Binary files a/doc/LectureNotes/_build/html/_images/week40_100_1.png and b/doc/LectureNotes/_build/html/_images/week40_100_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/week40_104_1.png b/doc/LectureNotes/_build/html/_images/week40_104_1.png index 7b6b79527..cda12710b 100644 Binary files a/doc/LectureNotes/_build/html/_images/week40_104_1.png and b/doc/LectureNotes/_build/html/_images/week40_104_1.png differ diff --git a/doc/LectureNotes/_build/html/_images/week40_34_1.png b/doc/LectureNotes/_build/html/_images/week40_34_1.png index fcf8aaf89..e3ffd3ea7 100644 Binary files a/doc/LectureNotes/_build/html/_images/week40_34_1.png and b/doc/LectureNotes/_build/html/_images/week40_34_1.png differ diff --git a/doc/LectureNotes/_build/html/_sources/exercisesweek42.ipynb b/doc/LectureNotes/_build/html/_sources/exercisesweek42.ipynb index 846880178..c6dd8e5a0 100644 --- a/doc/LectureNotes/_build/html/_sources/exercisesweek42.ipynb +++ b/doc/LectureNotes/_build/html/_sources/exercisesweek42.ipynb @@ -3,7 +3,9 @@ { "cell_type": "markdown", "id": "4b4c06bc", - "metadata": {}, + "metadata": { + "editable": true + }, "source": [ "\n", @@ -13,11 +15,13 @@ { "cell_type": "markdown", "id": "bcb25e64", - "metadata": {}, + "metadata": { + "editable": true + }, "source": [ "# Exercises week 42\n", "\n", - "**October 14-18, 2024**\n", + "**October 11-18, 2024**\n", "\n", "Date: **Deadline is Friday October 18 at midnight**\n" ] @@ -25,13 +29,17 @@ { "cell_type": "markdown", "id": "bb01f126", - "metadata": {}, + "metadata": { + "editable": true + }, "source": [ "# Overarching aims of the exercises this week\n", "\n", "The aim of the exercises this week is to get started with implementing a neural network. There are a lot of technical and finicky parts of implementing a neutal network, so take your time.\n", "\n", - "This week, you will implement only the feed-forward pass. Next week, you will implement backpropagation. We recommend that you do the exercises this week by editing and running this notebook file, as it includes several checks along the way that you have implemented the pieces of the feed-forward pass correctly. If you have trouble running a notebook, or importing pytorch, you can run this notebook in google colab instead: (LINK TO COLAB), though we recommend that you set up VSCode and your python environment to run code like this locally.\n" + "This week, you will implement only the feed-forward pass and updating the network parameters with simple gradient descent, the gradient will be computed using autograd using code we provide. Next week, you will implement backpropagation. We recommend that you do the exercises this week by editing and running this notebook file, as it includes some checks along the way that you have implemented the pieces of the feed-forward pass correctly, and running small parts of the code at a time will be important for understanding the methods.\n", + "\n", + "If you have trouble running a notebook, you can run this notebook in google colab instead (https://colab.research.google.com/drive/1OCQm1tlTWB6hZSf9I7gGUgW9M8SbVeQu#offline=true&sandboxMode=true), an updated link will be provided on the course discord (you can also send an email to k.h.fredly@fys.uio.no if you encounter any trouble), though we recommend that you set up VSCode and your python environment to run code like this locally.\n" ] }, { @@ -41,8 +49,33 @@ "metadata": {}, "outputs": [], "source": [ - "import autograd.numpy as np\n", - "from autograd import grad" + "import autograd.numpy as np # We need to use this numpy wrapper to make automatic differentiation work later\n", + "from sklearn import datasets\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.metrics import accuracy_score\n", + "\n", + "\n", + "# Defining some activation functions\n", + "def ReLU(z):\n", + " return np.where(z > 0, z, 0)\n", + "\n", + "\n", + "def sigmoid(z):\n", + " return 1 / (1 + np.exp(-z))\n", + "\n", + "\n", + "def softmax(z):\n", + " \"\"\"Compute softmax values for each set of scores in the rows of the matrix z.\n", + " Used with batched input data.\"\"\"\n", + " e_z = np.exp(z - np.max(z, axis=0))\n", + " return e_z / np.sum(e_z, axis=1)[:, np.newaxis]\n", + "\n", + "\n", + "def softmax_vec(z):\n", + " \"\"\"Compute softmax values for each set of scores in the vector z.\n", + " Use this function when you use the activation function on one vector at a time\"\"\"\n", + " e_z = np.exp(z - np.max(z))\n", + " return e_z / np.sum(e_z)" ] }, { @@ -52,7 +85,7 @@ "source": [ "# Exercise 1\n", "\n", - "Complete the following parts to compute the activation of the first layer.\n" + "In this exercise you will compute the activation of the first layer. You only need to change the code in the cells right below an exercise, the rest works out of the box. Feel free to make changes and see how stuff works though!\n" ] }, { @@ -64,21 +97,24 @@ "source": [ "np.random.seed(2024)\n", "\n", - "\n", - "def ReLU(z):\n", - " return np.where(z > 0, z, 0)\n", - "\n", - "\n", - "x = np.random.randn(2) # network input\n", + "x = np.random.randn(2) # network input. This is a single input with two features\n", "W1 = np.random.randn(4, 2) # first layer weights" ] }, + { + "cell_type": "markdown", + "id": "4ed2cf3d", + "metadata": {}, + "source": [ + "**a)** Given the shape of the first layer weight matrix, what is the input shape of the neural network? What is the output shape of the first layer?\n" + ] + }, { "cell_type": "markdown", "id": "edf7217b", "metadata": {}, "source": [ - "**a)** Define the bias of the first layer, `b1`with the correct shape\n" + "**b)** Define the bias of the first layer, `b1`with the correct shape. (Run the next cell right after the previous to get the random generated values to line up with the test solution below)\n" ] }, { @@ -88,7 +124,7 @@ "metadata": {}, "outputs": [], "source": [ - "b1 = np.random.randn(4)" + "b1 = ..." ] }, { @@ -96,7 +132,7 @@ "id": "09e8d453", "metadata": {}, "source": [ - "**b)** Compute the intermediary `z1` for the first layer\n" + "**c)** Compute the intermediary `z1` for the first layer\n" ] }, { @@ -106,7 +142,7 @@ "metadata": {}, "outputs": [], "source": [ - "z1 = W1 @ x + b1" + "z1 = ..." ] }, { @@ -114,7 +150,7 @@ "id": "6f71374e", "metadata": {}, "source": [ - "**c)** Compute the activation `a1` for the first layer using the ReLU activation function defined earlier.\n" + "**d)** Compute the activation `a1` for the first layer using the ReLU activation function defined earlier.\n" ] }, { @@ -124,7 +160,7 @@ "metadata": {}, "outputs": [], "source": [ - "a1 = ReLU(z1)" + "a1 = ..." ] }, { @@ -132,7 +168,7 @@ "id": "088710c0", "metadata": {}, "source": [ - "Confirm that you got the correct activation with the test below.\n" + "Confirm that you got the correct activation with the test below. Make sure that you define `b1` with the randn function right after you define `W1`.\n" ] }, { @@ -154,9 +190,11 @@ "source": [ "# Exercise 2\n", "\n", - "Compute the activation of the second layer with an output of length 8 and ReLU activation.\n", + "Now we will add a layer to the network with an output of length 8 and ReLU activation.\n", "\n", - "**a)** Define the weight and bias of the second layer with the right shapes.\n" + "**a)** What is the input of the second layer? What is its shape?\n", + "\n", + "**b)** Define the weight and bias of the second layer with the right shapes.\n" ] }, { @@ -166,8 +204,8 @@ "metadata": {}, "outputs": [], "source": [ - "W2 = np.random.randn(8, 4)\n", - "b2 = np.random.randn(8)" + "W2 = ...\n", + "b2 = ..." ] }, { @@ -175,7 +213,7 @@ "id": "5bd7d84b", "metadata": {}, "source": [ - "**b)** Compute intermediary `z2` and activation `a2` for the second layer.\n" + "**c)** Compute the intermediary `z2` and activation `a2` for the second layer.\n" ] }, { @@ -185,8 +223,8 @@ "metadata": {}, "outputs": [], "source": [ - "z2 = W2 @ a1\n", - "a2 = ReLU(z2)" + "z2 = ...\n", + "a2 = ..." ] }, { @@ -204,7 +242,9 @@ "metadata": {}, "outputs": [], "source": [ - "print(a2.shape == (8,))" + "print(\n", + " np.allclose(np.exp(len(a2)), 2980.9579870417283)\n", + ") # This should evaluate to True if a2 has the correct shape :)" ] }, { @@ -226,16 +266,16 @@ "metadata": {}, "outputs": [], "source": [ - "def create_layers(network_input_size, output_sizes):\n", + "def create_layers(network_input_size, layer_output_sizes):\n", " layers = []\n", "\n", " i_size = network_input_size\n", - " for output_size in output_sizes:\n", - " W = np.random.rand(output_size, i_size)\n", - " b = np.random.rand(output_size)\n", + " for layer_output_size in layer_output_sizes:\n", + " W = ...\n", + " b = ...\n", " layers.append((W, b))\n", "\n", - " i_size = output_size\n", + " i_size = layer_output_size\n", " return layers" ] }, @@ -244,7 +284,7 @@ "id": "bdc0cda2", "metadata": {}, "source": [ - "**b)** Comple the function below so that it evaluates the intermediate `z` and activation `a` for each layer, and returns the final activation `a`. This is the complete feed-forward pass, a full neural network!\n" + "**b)** Comple the function below so that it evaluates the intermediary `z` and activation `a` for each layer, with ReLU actication, and returns the final activation `a`. This is the complete feed-forward pass, a full neural network!\n" ] }, { @@ -254,11 +294,11 @@ "metadata": {}, "outputs": [], "source": [ - "def feed_forward(layers, input):\n", + "def feed_forward_all_relu(layers, input):\n", " a = input\n", " for W, b in layers:\n", - " z = W @ a + b\n", - " a = ReLU(z)\n", + " z = ...\n", + " a = ...\n", " return a" ] }, @@ -276,14 +316,30 @@ "id": "89a8f70d", "metadata": {}, "outputs": [], - "source": [] + "source": [ + "input_size = ...\n", + "layer_output_sizes = [...]\n", + "\n", + "x = np.random.rand(input_size)\n", + "layers = ...\n", + "predict = ...\n", + "print(predict)" + ] + }, + { + "cell_type": "markdown", + "id": "0da7fd52", + "metadata": {}, + "source": [ + "**d)** Why is a neural network with no activation functions always mathematically equivelent to a neural network with only one layer?\n" + ] }, { "cell_type": "markdown", "id": "306d8b7c", "metadata": {}, "source": [ - "# Exercise 4\n" + "# Exercise 4 - Custom activation for each layer\n" ] }, { @@ -291,29 +347,7 @@ "id": "221c7b6c", "metadata": {}, "source": [ - "So far, every layer has used the same activation, ReLU. We often want to use other types of activation however, so we need to update our code to support multiple types of activation. Make sure that you have completed every previous exercise before trying this one.\n", - "\n", - "**a)** Make the `create_layers` function also accept a list of activation functions, which is used to add activation functions to each of the tuples in `layers`. Make new functions to not mess with the old ones.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "id": "9df82312", - "metadata": {}, - "outputs": [], - "source": [ - "def create_layers_4(network_input_size, output_sizes, activation_funcs):\n", - " layers = []\n", - "\n", - " i_size = network_input_size\n", - " for output_size, activation in zip(output_sizes, activation_funcs):\n", - " W = np.random.rand(output_size, i_size)\n", - " b = np.random.rand(output_size)\n", - " layers.append((W, b, activation))\n", - "\n", - " i_size = output_size\n", - " return layers" + "So far, every layer has used the same activation, ReLU. We often want to use other types of activation however, so we need to update our code to support multiple types of activation functions. Make sure that you have completed every previous exercise before trying this one.\n" ] }, { @@ -321,7 +355,7 @@ "id": "10896d06", "metadata": {}, "source": [ - "**b)** Update the `feed_forward` function to support this change.\n" + "**a)** Complete the `feed_forward` function which accepts a list of activation functions as an argument, and which evaluates these activation functions at each layer.\n" ] }, { @@ -331,11 +365,109 @@ "metadata": {}, "outputs": [], "source": [ - "def feed_forward_4(layers, input):\n", + "def feed_forward(input, layers, activations):\n", " a = input\n", - " for W, b, activation in layers:\n", - " z = W @ a + b\n", - " a = activation(z)\n", + " for (W, b), activation in zip(layers, activations):\n", + " z = ...\n", + " a = ...\n", + " return a" + ] + }, + { + "cell_type": "markdown", + "id": "8f7df363", + "metadata": {}, + "source": [ + "**b)** Make a list with three activation functions(don't call them yet! you can make a list with function names as elements, and then call these elements of the list later), two ReLU and one sigmoid. (If you add other functions than the ones defined at the start of the notebook, make sure everything is defined using autograd's numpy wrapper, like above, since we want to use automatic differentiation on all of these functions later.)\n", + "\n", + "Then evaluate a network with three layers and these activation functions.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "301b46dc", + "metadata": {}, + "outputs": [], + "source": [ + "network_input_size = ...\n", + "layer_output_sizes = [...]\n", + "activations = [...]\n", + "layers = ...\n", + "\n", + "x = np.random.randn(network_input_size)\n", + "feed_forward(x, layers, activations)" + ] + }, + { + "cell_type": "markdown", + "id": "a8d6c425", + "metadata": {}, + "source": [ + "# Exercise 5 - Processing multiple inputs at once\n" + ] + }, + { + "cell_type": "markdown", + "id": "0f4330a4", + "metadata": {}, + "source": [ + "So far, the feed forward function has taken one input vector as an input. This vector then undergoes a linear transformation and then an element-wise non-linear operation for each layer. This approach of sending one vector in at a time is great for interpreting how the network transforms data with its linear and non-linear operations, but not the best for numerical efficiency. Now, we want to be able to send many inputs through the network at once. This will make the code a bit harder to understand, but it will make it faster, and more compact. It will be worth the trouble.\n", + "\n", + "To process multiple inputs at once, while still performing the same operations, you will only need to flip a couple things around.\n" + ] + }, + { + "cell_type": "markdown", + "id": "17023bb7", + "metadata": {}, + "source": [ + "**a)** Complete the function `create_layers_batch` so that the weight matrix is the transpose of what it was when you only sent in one input at a time.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "a241fd79", + "metadata": {}, + "outputs": [], + "source": [ + "def create_layers_batch(network_input_size, layer_output_sizes):\n", + " layers = []\n", + "\n", + " i_size = network_input_size\n", + " for layer_output_size in layer_output_sizes:\n", + " W = ...\n", + " b = ...\n", + " layers.append((W, b))\n", + "\n", + " i_size = layer_output_size\n", + " return layers" + ] + }, + { + "cell_type": "markdown", + "id": "a6349db6", + "metadata": {}, + "source": [ + "**b)** Make a matrix of inputs with the shape (number of features, number of inputs), you choose the number of inputs and features per input. Then complete the function `feed_forward_batch` so that you can process this matrix of inputs with only one matrix multiplication and one broadcasted vector addition per layer. (Hint: You will only need to swap two variable around from your previous implementation, but remember to test that you get the same results for equivelent inputs!)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "425f3bcc", + "metadata": {}, + "outputs": [], + "source": [ + "inputs = np.random.rand(1000, 4)\n", + "\n", + "\n", + "def feed_forward_batch(inputs, layers, activations):\n", + " a = inputs\n", + " for (W, b), activation in zip(layers, activations):\n", + " z = ...\n", + " a = ...\n", " return a" ] }, @@ -354,15 +486,29 @@ "metadata": {}, "outputs": [], "source": [ - "from scipy.special import softmax\n", - "\n", - "network_input_size = 4\n", - "output_sizes = [12, 10, 3]\n", - "activation_funcs = [ReLU, ReLU, softmax]\n", - "layers = create_layers_4(network_input_size, output_sizes, activation_funcs)\n", + "network_input_size = ...\n", + "layer_output_sizes = [...]\n", + "activations = [...]\n", + "layers = create_layers_batch(network_input_size, layer_output_sizes)\n", "\n", "x = np.random.randn(network_input_size)\n", - "predict = feed_forward_4(layers, x)" + "feed_forward_batch(inputs, layers, activations)" + ] + }, + { + "cell_type": "markdown", + "id": "87999271", + "metadata": {}, + "source": [ + "You should use this batched approach moving forward, as it will lead to much more compact code. However, remember that each input is still treated separately, and that you will need to keep in mind the transposed weight matrix and other details when implementing backpropagation.\n" + ] + }, + { + "cell_type": "markdown", + "id": "237eb782", + "metadata": {}, + "source": [ + "# Exercise 6 - Predicting on real data\n" ] }, { @@ -370,9 +516,9 @@ "id": "54d5fde7", "metadata": {}, "source": [ - "The final exercise will hopefully be very simple if everything has worked so far. You will evaluate your neural network on the iris data set (https://scikit-learn.org/1.5/auto_examples/datasets/plot_iris_dataset.html).\n", + "You will now evaluate your neural network on the iris data set (https://scikit-learn.org/1.5/auto_examples/datasets/plot_iris_dataset.html).\n", "\n", - "This dataset contains data on 150 flowers of 3 different types which can be separated pretty well using the four features given for each flower, which includes the width and length of their leaves. You are not expected to do any training of the network or actual classification, unless you feel like it, in that case you can do exercise 5.\n" + "This dataset contains data on 150 flowers of 3 different types which can be separated pretty well using the four features given for each flower, which includes the width and length of their leaves. You are will later train your network to actually make good predictions.\n" ] }, { @@ -382,10 +528,6 @@ "metadata": {}, "outputs": [], "source": [ - "# Loading and plotting iris dataset\n", - "from sklearn import datasets\n", - "import matplotlib.pyplot as plt\n", - "\n", "iris = datasets.load_iris()\n", "\n", "_, ax = plt.subplots()\n", @@ -396,24 +538,91 @@ ")" ] }, + { + "cell_type": "code", + "execution_count": null, + "id": "ed3e2fc9", + "metadata": {}, + "outputs": [], + "source": [ + "inputs = iris.data\n", + "\n", + "# Since each prediction is a vector with a score for each of the three types of flowers,\n", + "# we need to make each target a vector with a 1 for the correct flower and a 0 for the others.\n", + "targets = np.zeros((len(iris.data), 3))\n", + "for i, t in enumerate(iris.target):\n", + " targets[i, t] = 1\n", + "\n", + "\n", + "def accuracy(predictions, targets):\n", + " one_hot_predictions = np.zeros(predictions.shape)\n", + "\n", + " for i, prediction in enumerate(predictions):\n", + " one_hot_predictions[i, np.argmax(prediction)] = 1\n", + " return accuracy_score(one_hot_predictions, targets)" + ] + }, { "cell_type": "markdown", - "id": "c528846f", + "id": "0362c4a9", "metadata": {}, "source": [ - "**c)** Loop over the iris dataset(`iris.data`) and evaluate the network for each data point.\n" + "**a)** What should the input size for the network be with this dataset? What should the output shape of the last layer be?\n" + ] + }, + { + "cell_type": "markdown", + "id": "bf62607e", + "metadata": {}, + "source": [ + "**b)** Create a network with two hidden layers, the first with sigmoid activation and the last with softmax, the first layer should have 8 \"nodes\", the second has the number of nodes you found in exercise a). Softmax returns a \"probability distribution\", in the sense that the numbers in the output are positive and add up to 1 and, their magnitude are in some sense relative to their magnitude before going through the softmax function. Remember to use the batched version of the create_layers and feed forward functions.\n" ] }, { "cell_type": "code", "execution_count": null, - "id": "2efc507d", + "id": "5366d4ae", "metadata": {}, "outputs": [], "source": [ - "# No need to change this cell! Just make sure it works!\n", - "for x in iris.data:\n", - " prediction = feed_forward_4(layers, x)" + "...\n", + "layers = ..." + ] + }, + { + "cell_type": "markdown", + "id": "c528846f", + "metadata": {}, + "source": [ + "**c)** Evaluate your model on the entire iris dataset! For later purposes, we will split the data into train and test sets, and compute gradients on smaller batches of the training data. But for now, evaluate the network on the whole thing at once.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6c783105", + "metadata": {}, + "outputs": [], + "source": [ + "predictions = feed_forward_batch(inputs, layers, activations)" + ] + }, + { + "cell_type": "markdown", + "id": "01a3caa8", + "metadata": {}, + "source": [ + "**d)** Compute the accuracy of your model using the accuracy function defined above. Recreate your model a couple times and see how the accuracy changes.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "a2612b82", + "metadata": {}, + "outputs": [], + "source": [ + "print(accuracy(predictions, targets))" ] }, { @@ -421,31 +630,126 @@ "id": "334560b6", "metadata": {}, "source": [ - "# Exercise 5 (Very optional and very hard :)\n" + "# Exercise 6 - Training on real data\n", + "\n", + "To be able to actually do anything useful with your neural network, you need to train it. For this, we need a cost function and a way to take the gradient of the cost function wrt. the network parameters. The following exercises guide you through taking the gradient using autograd, and updating the network parameters using the gradient. Feel free to implement gradient methods like ADAM if you finish everything.\n" ] }, { "cell_type": "markdown", - "id": "ea0a8fe0", + "id": "700cabe4", "metadata": {}, "source": [ - "**a)** Make the iris target values into one-hot vectors.\n", + "The cross-entropy loss function can evaluate performance on classification tasks. It sees if your prediction is \"most certain\" on the correct target.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "56bef776", + "metadata": {}, + "outputs": [], + "source": [ + "from autograd import grad\n", "\n", - "**b)** Define the cross-entropy loss function to evaluate the performance of your network on the data set.\n", "\n", - "**c)** Use the autograd package to take the gradient of the cross entropy wrt. the weights and biases of the network.\n", + "def cost(input, layers, activations, target):\n", + " predict = feed_forward_batch(input, layers, activations)\n", + " return cross_entropy(predict, target)\n", "\n", - "**d)** Use gradient descent of some sort to optimize the parameters.\n", "\n", - "**e)** Evaluate the accuracy of the network.\n", + "def cross_entropy(predict, target):\n", + " return np.sum(-target * np.log(predict))\n", "\n", - "**e)** Show off how you did in a group session!\n" + "\n", + "gradient_func = grad(\n", + " cross_entropy, 1\n", + ") # Taking the gradient wrt. the second input to the cost function" + ] + }, + { + "cell_type": "markdown", + "id": "7b1b74bc", + "metadata": {}, + "source": [ + "**a)** What shape should the gradient of the cost function wrt. weights and biases be?\n", + "\n", + "**b)** Use the `gradient_func` function to take the gradient of the cross entropy wrt. the weights and biases of the network. Check the shapes of what's inside. What does the `grad` func from autograd actually do?\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "841c9e87", + "metadata": {}, + "outputs": [], + "source": [ + "layers_grad = gradient_func(inputs, layers, activations, targets) # Don't change this" + ] + }, + { + "cell_type": "markdown", + "id": "adc9e9be", + "metadata": {}, + "source": [ + "**c)** Finish the `train_network` function.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6e4d38d3", + "metadata": {}, + "outputs": [], + "source": [ + "def train_network(\n", + " inputs, layers, activations, targets, learning_rate=0.001, epochs=100\n", + "):\n", + " for i in range(epochs):\n", + " layers_grad = gradient_func(inputs, layers, activations, targets)\n", + " for (W, b), (W_g, b_g) in zip(layers, layers_grad):\n", + " W -= ...\n", + " b -= ..." + ] + }, + { + "cell_type": "markdown", + "id": "2f65d663", + "metadata": {}, + "source": [ + "**e)** What do we call the gradient method used above?\n" + ] + }, + { + "cell_type": "markdown", + "id": "7059dd8c", + "metadata": {}, + "source": [ + "**d)** Train your network and see how the accuracy changes! Make a plot if you want.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "5027c7a5", + "metadata": {}, + "outputs": [], + "source": [ + "..." + ] + }, + { + "cell_type": "markdown", + "id": "3bc77016", + "metadata": {}, + "source": [ + "**e)** How high of an accuracy is it possible to acheive with a neural network on this dataset, if we use the whole thing as training data?\n" ] } ], "metadata": { "kernelspec": { - "display_name": "Python 3 (ipykernel)", + "display_name": ".venv", "language": "python", "name": "python3" }, @@ -459,7 +763,7 @@ "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", - "version": "3.9.18" + "version": "3.12.7" } }, "nbformat": 4, diff --git a/doc/LectureNotes/_build/html/chapter1.html b/doc/LectureNotes/_build/html/chapter1.html index 82b934576..dbdf858f3 100644 --- a/doc/LectureNotes/_build/html/chapter1.html +++ b/doc/LectureNotes/_build/html/chapter1.html @@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output" Week 41 Neural networks and constructing a neural network code +
@@ -1056,13 +1066,13 @@ example of the functionality of Scikit-Learn.
The intercept alpha:
- [1.94485679]
+ [2.07549007]
Coefficient beta :
- [[5.13059483]]
-Mean squared error: 0.32
-Variance score: 0.87
+ [[5.14029264]]
+Mean squared error: 0.23
+Variance score: 0.89
Mean squared log error: 0.01
-Mean absolute error: 0.45
+Mean absolute error: 0.38
@@ -1162,7 +1172,7 @@ a linear \(x\)-dependence we s
-0.004999999999999996
+0.004999999999999987
@@ -1373,7 +1383,7 @@ the Hadamard product, meaning element-wise multiplication.
Old accuracy on training data: 0.1440501043841336
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1707,7 +1717,7 @@ Lambda = 10.0
Accuracy score on test set: 0.19166666666666668
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1716,7 +1726,7 @@ Lambda = 1e-05
Accuracy score on test set: 0.10555555555555556
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1725,7 +1735,7 @@ Lambda = 0.0001
Accuracy score on test set: 0.08611111111111111
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1734,7 +1744,7 @@ Lambda = 0.001
Accuracy score on test set: 0.10555555555555556
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1743,7 +1753,7 @@ Lambda = 0.01
Accuracy score on test set: 0.08888888888888889
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1752,7 +1762,7 @@ Lambda = 0.1
Accuracy score on test set: 0.08611111111111111
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
@@ -1761,191 +1771,34 @@ Lambda = 1.0
Accuracy score on test set: 0.08888888888888889
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57122/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-Learning rate = 0.1
-Lambda = 10.0
-Accuracy score on test set: 0.09166666666666666
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 1.0
-Lambda = 1e-05
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 1.0
-Lambda = 0.0001
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 1.0
-Lambda = 0.001
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 1.0
-Lambda = 0.01
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 1.0
-Lambda = 0.1
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-
-
-Learning rate = 1.0
-Lambda = 1.0
-Accuracy score on test set: 0.10555555555555556
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 1.0
-Lambda = 10.0
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 1e-05
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 0.0001
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 0.001
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 0.01
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 0.1
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 1.0
-Accuracy score on test set: 0.07777777777777778
-
-
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:43: RuntimeWarning: overflow encountered in exp
- exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
- self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
-
-
-Learning rate = 10.0
-Lambda = 10.0
-Accuracy score on test set: 0.07777777777777778
+---------------------------------------------------------------------------
+KeyboardInterrupt Traceback (most recent call last)
+Cell In[8], line 11
+ 8 for j, lmbd in enumerate(lmbd_vals):
+ 9 dnn = NeuralNetwork(X_train, Y_train_onehot, eta=eta, lmbd=lmbd, epochs=epochs, batch_size=batch_size,
+ 10 n_hidden_neurons=n_hidden_neurons, n_categories=n_categories)
+---> 11 dnn.train()
+ 13 DNN_numpy[i][j] = dnn
+ 15 test_predict = dnn.predict(X_test)
+
+Cell In[6], line 98, in NeuralNetwork.train(self)
+ 95 self.X_data = self.X_data_full[chosen_datapoints]
+ 96 self.Y_data = self.Y_data_full[chosen_datapoints]
+---> 98 self.feed_forward()
+ 99 self.backpropagation()
+
+Cell In[6], line 38, in NeuralNetwork.feed_forward(self)
+ 36 def feed_forward(self):
+ 37 # feed-forward for training
+---> 38 self.z_h = np.matmul(self.X_data, self.hidden_weights) + self.hidden_bias
+ 39 self.a_h = sigmoid(self.z_h)
+ 41 self.z_o = np.matmul(self.a_h, self.output_weights) + self.output_bias
+
+KeyboardInterrupt:
@@ -1991,22 +1844,6 @@ Accuracy score on test set: 0.07777777777777778
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22411/953065564.py:4: RuntimeWarning: overflow encountered in exp
- return 1/(1 + np.exp(-x))
-
-
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 1e-05
-Accuracy score on test set: 0.18333333333333332
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 0.0001
-Accuracy score on test set: 0.18611111111111112
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 0.001
-Accuracy score on test set: 0.13055555555555556
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 0.01
-Accuracy score on test set: 0.24444444444444444
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 0.1
-Accuracy score on test set: 0.23333333333333334
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 1.0
-Accuracy score on test set: 0.12777777777777777
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 1e-05
-Lambda = 10.0
-Accuracy score on test set: 0.1527777777777778
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 1e-05
-Accuracy score on test set: 0.9111111111111111
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 0.0001
-Accuracy score on test set: 0.8888888888888888
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 0.001
-Accuracy score on test set: 0.8722222222222222
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 0.01
-Accuracy score on test set: 0.8305555555555556
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 0.1
-Accuracy score on test set: 0.8888888888888888
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 1.0
-Accuracy score on test set: 0.8805555555555555
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.0001
-Lambda = 10.0
-Accuracy score on test set: 0.8944444444444445
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 1e-05
-Accuracy score on test set: 0.975
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 0.0001
-Accuracy score on test set: 0.9777777777777777
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 0.001
-Accuracy score on test set: 0.9805555555555555
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 0.01
-Accuracy score on test set: 0.9861111111111112
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 0.1
-Accuracy score on test set: 0.9805555555555555
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 1.0
-Accuracy score on test set: 0.9777777777777777
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.001
-Lambda = 10.0
-Accuracy score on test set: 0.9444444444444444
-/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/neural_network/_multilayer_perceptron.py:691: ConvergenceWarning: Stochastic Optimizer: Maximum iterations (100) reached and the optimization hasn't converged yet.
- warnings.warn(
-Learning rate = 0.01
-Lambda = 1e-05
-Accuracy score on test set: 0.9861111111111112
-Learning rate = 0.01
-Lambda = 0.0001
-Accuracy score on test set: 0.9888888888888889
-Learning rate = 0.01
-Lambda = 0.001
-Accuracy score on test set: 0.9888888888888889
-Learning rate = 0.01
-Lambda = 0.01
-Accuracy score on test set: 0.9861111111111112
-Learning rate = 0.01
-Lambda = 0.1
-Accuracy score on test set: 0.9888888888888889
-Learning rate = 0.01
-Lambda = 1.0
-Accuracy score on test set: 0.9722222222222222
-
-Learning rate = 0.01
-Lambda = 10.0
-Accuracy score on test set: 0.9527777777777777
-Learning rate = 0.1
-Lambda = 1e-05
-Accuracy score on test set: 0.9027777777777778
-Learning rate = 0.1
-Lambda = 0.0001
-Accuracy score on test set: 0.8583333333333333
-Learning rate = 0.1
-Lambda = 0.001
-Accuracy score on test set: 0.8722222222222222
-Learning rate = 0.1
-Lambda = 0.01
-Accuracy score on test set: 0.9055555555555556
-Learning rate = 0.1
-Lambda = 0.1
-Accuracy score on test set: 0.8805555555555555
-Learning rate = 0.1
-Lambda = 1.0
-Accuracy score on test set: 0.8722222222222222
-Learning rate = 0.1
-Lambda = 10.0
-Accuracy score on test set: 0.8666666666666667
-Learning rate = 1.0
-Lambda = 1e-05
-Accuracy score on test set: 0.08611111111111111
-
-Learning rate = 1.0
-Lambda = 0.0001
-Accuracy score on test set: 0.10555555555555556
-Learning rate = 1.0
-Lambda = 0.001
-Accuracy score on test set: 0.10555555555555556
-
-Learning rate = 1.0
-Lambda = 0.01
-Accuracy score on test set: 0.17777777777777778
-Learning rate = 1.0
-Lambda = 0.1
-Accuracy score on test set: 0.08333333333333333
-
-Learning rate = 1.0
-Lambda = 1.0
-Accuracy score on test set: 0.08888888888888889
-Learning rate = 1.0
-Lambda = 10.0
-Accuracy score on test set: 0.09444444444444444
-
-Learning rate = 10.0
-Lambda = 1e-05
-Accuracy score on test set: 0.17222222222222222
-Learning rate = 10.0
-Lambda = 0.0001
-Accuracy score on test set: 0.11666666666666667
-
-Learning rate = 10.0
-Lambda = 0.001
-Accuracy score on test set: 0.10555555555555556
-Learning rate = 10.0
-Lambda = 0.01
-Accuracy score on test set: 0.1388888888888889
-
-Learning rate = 10.0
-Lambda = 0.1
-Accuracy score on test set: 0.11388888888888889
-Learning rate = 10.0
-Lambda = 1.0
-Accuracy score on test set: 0.10555555555555556
-
-Learning rate = 10.0
-Lambda = 10.0
-Accuracy score on test set: 0.09444444444444444
-
-
- Cell In[12], line 1
- conda create -n tf tensorflow
- ^
-SyntaxError: invalid syntax
-To install the current release of GPU TensorFlow
@@ -2662,6 +2672,43 @@ Using TensorFlow results in a much better execution time. Try it!
19 x = tuple(args[i] for i in argnum) ---> 20 return unary_operator(unary_f, x, *nary_op_args, **nary_op_kwargs) +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/differential_operators.py:60, in jacobian(fun, x) + 50 @unary_to_nary + 51 def jacobian(fun, x): + 52 """ + 53 Returns a function which computes the Jacobian of `fun` with respect to + 54 positional argument number `argnum`, which must be a scalar or array. Unlike + (...) + 58 (out1, out2, ...) then the Jacobian has shape (out1, out2, ..., in1, in2, ...). + 59 """ +---> 60 vjp, ans = _make_vjp(fun, x) + 61 ans_vspace = vspace(ans) + 62 jacobian_shape = ans_vspace.shape + vspace(x).shape + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/core.py:10, in make_vjp(fun, x) + 8 def make_vjp(fun, x): + 9 start_node = VJPNode.new_root() +---> 10 end_value, end_node = trace(start_node, fun, x) + 11 if end_node is None: + 12 def vjp(g): return vspace(x).zeros() + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/tracer.py:10, in trace(start_node, fun, x) + 8 with trace_stack.new_trace() as t: + 9 start_box = new_box(x, t, start_node) +---> 10 end_box = fun(start_box) + 11 if isbox(end_box) and end_box._trace == start_box._trace: + 12 return end_box._value, end_box._node + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/wrap_util.py:15, in unary_to_nary.<locals>.nary_operator.<locals>.nary_f.<locals>.unary_f(x) + 13 else: + 14 subargs = subvals(args, zip(argnum, x)) +---> 15 return fun(*subargs, **kwargs) + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/wrap_util.py:20, in unary_to_nary.<locals>.nary_operator.<locals>.nary_f(*args, **kwargs) + 18 else: + 19 x = tuple(args[i] for i in argnum) +---> 20 return unary_operator(unary_f, x, *nary_op_args, **nary_op_kwargs) + File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/differential_operators.py:64, in jacobian(fun, x) 62 jacobian_shape = ans_vspace.shape + vspace(x).shape 63 grads = map(vjp, ans_vspace.standard_basis()) @@ -2695,43 +2742,58 @@ Using TensorFlow results in a much better execution time. Try it! 22 for parent, ingrad in zip(node.parents, ingrads): 23 outgrads[parent] = add_outgrads(outgrads.get(parent), ingrad) -File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/core.py:78, in defvjp.<locals>.vjp_argnums.<locals>.<lambda>(g) - 76 vjp_0 = vjp_0_fun(ans, *args, **kwargs) - 77 vjp_1 = vjp_1_fun(ans, *args, **kwargs) ----> 78 return lambda g: (vjp_0(g), vjp_1(g)) - 79 else: - 80 vjps = [vjps_dict[argnum](ans, *args, **kwargs) for argnum in argnums] +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/core.py:67, in defvjp.<locals>.vjp_argnums.<locals>.<lambda>(g) + 64 raise NotImplementedError( + 65 "VJP of {} wrt argnum 0 not defined".format(fun.__name__)) + 66 vjp = vjpfun(ans, *args, **kwargs) +---> 67 return lambda g: (vjp(g),) + 68 elif L == 2: + 69 argnum_0, argnum_1 = argnums -File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_vjps.py:660, in unbroadcast_f.<locals>.<lambda>(g) - 658 def unbroadcast_f(target, f): - 659 target_meta = anp.metadata(target) ---> 660 return lambda g: unbroadcast(f(g), target_meta) +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_vjps.py:423, in matmul_vjp_1.<locals>.<lambda>(g) + 421 A_ndim = anp.ndim(A) + 422 B_meta = anp.metadata(B) +--> 423 return lambda g: matmul_adjoint_1(A, g, A_ndim, B_meta) -File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_vjps.py:34, in <lambda>(g) - 30 # ----- Binary ufuncs ----- - 32 defvjp(anp.add, lambda ans, x, y : unbroadcast_f(x, lambda g: g), - 33 lambda ans, x, y : unbroadcast_f(y, lambda g: g)) ----> 34 defvjp(anp.multiply, lambda ans, x, y : unbroadcast_f(x, lambda g: y * g), - 35 lambda ans, x, y : unbroadcast_f(y, lambda g: x * g)) - 36 defvjp(anp.subtract, lambda ans, x, y : unbroadcast_f(x, lambda g: g), - 37 lambda ans, x, y : unbroadcast_f(y, lambda g: -g)) - 38 defvjp(anp.divide, lambda ans, x, y : unbroadcast_f(x, lambda g: g / y), - 39 lambda ans, x, y : unbroadcast_f(y, lambda g: - g * x / y**2)) +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_vjps.py:413, in matmul_adjoint_1(A, G, A_ndim, B_meta) + 411 if B_is_vec: + 412 result = anp.squeeze(result, anp.ndim(G) - 1) +--> 413 return unbroadcast(result, B_meta) -File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_boxes.py:27, in ArrayBox.__mul__(self, other) ----> 27 def __mul__(self, other): return anp.multiply(self, other) +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_vjps.py:653, in unbroadcast(x, target_meta, broadcast_idx) + 651 for axis, size in enumerate(target_shape): + 652 if size == 1: +--> 653 x = anp.sum(x, axis=axis, keepdims=True) + 654 if anp.iscomplexobj(x) and not target_iscomplex: + 655 x = anp.real(x) -File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/tracer.py:44, in primitive.<locals>.f_wrapped(*args, **kwargs) - 42 parents = tuple(box._node for _ , box in boxed_args) - 43 argnums = tuple(argnum for argnum, _ in boxed_args) ----> 44 ans = f_wrapped(*argvals, **kwargs) - 45 node = node_constructor(ans, f_wrapped, argvals, kwargs, argnums, parents) - 46 return new_box(ans, trace, node) - -File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/tracer.py:48, in primitive.<locals>.f_wrapped(*args, **kwargs) +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/tracer.py:45, in primitive.<locals>.f_wrapped(*args, **kwargs) + 43 argnums = tuple(argnum for argnum, _ in boxed_args) + 44 ans = f_wrapped(*argvals, **kwargs) +---> 45 node = node_constructor(ans, f_wrapped, argvals, kwargs, argnums, parents) 46 return new_box(ans, trace, node) 47 else: ----> 48 return f_raw(*args, **kwargs) + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/core.py:36, in VJPNode.__init__(self, value, fun, args, kwargs, parent_argnums, parents) + 33 fun_name = getattr(fun, '__name__', fun) + 34 raise NotImplementedError("VJP of {} wrt argnums {} not defined" + 35 .format(fun_name, parent_argnums)) +---> 36 self.vjp = vjpmaker(parent_argnums, value, args, kwargs) + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/core.py:66, in defvjp.<locals>.vjp_argnums(argnums, ans, args, kwargs) + 63 except KeyError: + 64 raise NotImplementedError( + 65 "VJP of {} wrt argnum 0 not defined".format(fun.__name__)) +---> 66 vjp = vjpfun(ans, *args, **kwargs) + 67 return lambda g: (vjp(g),) + 68 elif L == 2: + +File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/numpy/numpy_vjps.py:297, in grad_np_sum(ans, x, axis, keepdims, dtype) + 294 return lambda g: anp.sum(g, axis=broadcast_axes, keepdims=True) + 295 defvjp(anp.broadcast_to, grad_broadcast_to) +--> 297 def grad_np_sum(ans, x, axis=None, keepdims=False, dtype=None): + 298 shape, dtype = anp.shape(x), anp.result_type(x) + 299 return lambda g: repeat_to_match_shape(g, shape, dtype, axis, keepdims)[0] KeyboardInterrupt:
@@ -1235,12 +1245,12 @@ labels = (n_inputs) = (1797,)
2 from tensorflow.keras.layers import Input
3 from tensorflow.keras.models import Sequential #This allows appending layers to existing models
-File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/tensorflow/__init__.py:443
- 441 _plugin_dir = _os.path.join(_s, 'tensorflow-plugins')
- 442 if _os.path.exists(_plugin_dir):
---> 443 _ll.load_library(_plugin_dir)
- 444 # Load Pluggable Device Library
- 445 _ll.load_pluggable_device_library(_plugin_dir)
+File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/tensorflow/__init__.py:440
+ 438 _plugin_dir = _os.path.join(_s, 'tensorflow-plugins')
+ 439 if _os.path.exists(_plugin_dir):
+--> 440 _ll.load_library(_plugin_dir)
+ 441 # Load Pluggable Device Library
+ 442 _ll.load_pluggable_device_library(_plugin_dir)
File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/tensorflow/python/framework/load_library.py:151, in load_library(library_location)
148 kernel_libraries = [library_location]
diff --git a/doc/LectureNotes/_build/html/chapter13.html b/doc/LectureNotes/_build/html/chapter13.html
index 3828cc5ae..c80a1f73e 100644
--- a/doc/LectureNotes/_build/html/chapter13.html
+++ b/doc/LectureNotes/_build/html/chapter13.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
@@ -647,12 +657,12 @@ systems such as automatic translation and speech-to-text.
@@ -1305,10 +1315,10 @@ covariance matrix through the np.linalg.eig() function. We see here that we reach a plateau for the Ridge results. Writing out the coefficients \(\boldsymbol{\beta}\), we observe that they are getting smaller and smaller and our error stabilizes since the predicted values of \(\tilde{\boldsymbol{y}}\) approach zero.
@@ -859,10 +869,10 @@ number \(i\) is left out. Usin
The bias-variance tradeoff summarizes the fundamental tension in
@@ -1647,12 +1661,12 @@ Mean squared error on test data: 877.21517262
Degree of polynomial: 23
Mean squared error on training data: 0.00085892
Mean squared error on test data: 5567.04664255
-Degree of polynomial: 24
-Mean squared error on training data: 0.00084707
-Mean squared error on test data: 1325.26124692
-
diff --git a/doc/LectureNotes/_build/html/chapter5.html b/doc/LectureNotes/_build/html/chapter5.html
index 394c0c69a..1fac69acc 100644
--- a/doc/LectureNotes/_build/html/chapter5.html
+++ b/doc/LectureNotes/_build/html/chapter5.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/chapter6.html b/doc/LectureNotes/_build/html/chapter6.html
index 9c904f401..5e85b0cbf 100644
--- a/doc/LectureNotes/_build/html/chapter6.html
+++ b/doc/LectureNotes/_build/html/chapter6.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
@@ -787,9 +797,9 @@ predicting the target features of query instances is as follows:
diff --git a/doc/LectureNotes/_build/html/chapter8.html b/doc/LectureNotes/_build/html/chapter8.html
index d85525649..a38643236 100644
--- a/doc/LectureNotes/_build/html/chapter8.html
+++ b/doc/LectureNotes/_build/html/chapter8.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
@@ -741,10 +751,10 @@ covariance matrix through the np.linalg.eig() function.
diff --git a/doc/LectureNotes/_build/html/chapteroptimization.html b/doc/LectureNotes/_build/html/chapteroptimization.html
index 637c870f3..838dfe57c 100644
--- a/doc/LectureNotes/_build/html/chapteroptimization.html
+++ b/doc/LectureNotes/_build/html/chapteroptimization.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/clustering.html b/doc/LectureNotes/_build/html/clustering.html
index ee79e09c1..44e1749ec 100644
--- a/doc/LectureNotes/_build/html/clustering.html
+++ b/doc/LectureNotes/_build/html/clustering.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
@@ -600,12 +610,12 @@ Gaussian distribution.
diff --git a/doc/LectureNotes/_build/html/exercisesweek35.html b/doc/LectureNotes/_build/html/exercisesweek35.html
index 2f0422528..552519e48 100644
--- a/doc/LectureNotes/_build/html/exercisesweek35.html
+++ b/doc/LectureNotes/_build/html/exercisesweek35.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/exercisesweek36.html b/doc/LectureNotes/_build/html/exercisesweek36.html
index b2187bd49..8f3f40509 100644
--- a/doc/LectureNotes/_build/html/exercisesweek36.html
+++ b/doc/LectureNotes/_build/html/exercisesweek36.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/exercisesweek37.html b/doc/LectureNotes/_build/html/exercisesweek37.html
index f6a46f580..b9223bc23 100644
--- a/doc/LectureNotes/_build/html/exercisesweek37.html
+++ b/doc/LectureNotes/_build/html/exercisesweek37.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/exercisesweek38.html b/doc/LectureNotes/_build/html/exercisesweek38.html
index 8bea48d90..80ddafe25 100644
--- a/doc/LectureNotes/_build/html/exercisesweek38.html
+++ b/doc/LectureNotes/_build/html/exercisesweek38.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/exercisesweek39.html b/doc/LectureNotes/_build/html/exercisesweek39.html
index 74bb4cb9b..fa3740090 100644
--- a/doc/LectureNotes/_build/html/exercisesweek39.html
+++ b/doc/LectureNotes/_build/html/exercisesweek39.html
@@ -321,6 +321,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+
diff --git a/doc/LectureNotes/_build/html/exercisesweek41.html b/doc/LectureNotes/_build/html/exercisesweek41.html
index 6ae56b051..39e5531ee 100644
--- a/doc/LectureNotes/_build/html/exercisesweek41.html
+++ b/doc/LectureNotes/_build/html/exercisesweek41.html
@@ -804,15 +804,15 @@ regression. October 14-18, 2024 October 11-18, 2024 Date: Deadline is Friday October 18 at midnight The aim of the exercises this week is to get started with implementing a neural network. There are a lot of technical and finicky parts of implementing a neutal network, so take your time. This week, you will implement only the feed-forward pass. Next week, you will implement backpropagation. We recommend that you do the exercises this week by editing and running this notebook file, as it includes several checks along the way that you have implemented the pieces of the feed-forward pass correctly. If you have trouble running a notebook, or importing pytorch, you can run this notebook in google colab instead: (LINK TO COLAB), though we recommend that you set up VSCode and your python environment to run code like this locally. This week, you will implement only the feed-forward pass and updating the network parameters with simple gradient descent, the gradient will be computed using autograd using code we provide. Next week, you will implement backpropagation. We recommend that you do the exercises this week by editing and running this notebook file, as it includes some checks along the way that you have implemented the pieces of the feed-forward pass correctly, and running small parts of the code at a time will be important for understanding the methods. If you have trouble running a notebook, you can run this notebook in google colab instead (https://colab.research.google.com/drive/1OCQm1tlTWB6hZSf9I7gGUgW9M8SbVeQu#offline=true&sandboxMode=true), an updated link will be provided on the course discord (you can also send an email to k.h.fredly@fys.uio.no if you encounter any trouble), though we recommend that you set up VSCode and your python environment to run code like this locally. Complete the following parts to compute the activation of the first layer. In this exercise you will compute the activation of the first layer. You only need to change the code in the cells right below an exercise, the rest works out of the box. Feel free to make changes and see how stuff works though! a) Define the bias of the first layer, a) Given the shape of the first layer weight matrix, what is the input shape of the neural network? What is the output shape of the first layer? b) Define the bias of the first layer, b) Compute the intermediary c) Compute the intermediary c) Compute the activation d) Compute the activation Confirm that you got the correct activation with the test below. Confirm that you got the correct activation with the test below. Make sure that you define Compute the activation of the second layer with an output of length 8 and ReLU activation. a) Define the weight and bias of the second layer with the right shapes. Now we will add a layer to the network with an output of length 8 and ReLU activation. a) What is the input of the second layer? What is its shape? b) Define the weight and bias of the second layer with the right shapes. b) Compute intermediary c) Compute the intermediary Confirm that you got the correct activation shape with the test below. a) Complete the function below so that it returns a list b) Comple the function below so that it evaluates the intermediate b) Comple the function below so that it evaluates the intermediary c) Create a network with input size 8 and layers with output sizes 10, 16, 6, 2. Evaluate it and make sure that you get the correct size vectors along the way. So far, every layer has used the same activation, ReLU. We often want to use other types of activation however, so we need to update our code to support multiple types of activation. Make sure that you have completed every previous exercise before trying this one. a) Make the d) Why is a neural network with no activation functions always mathematically equivelent to a neural network with only one layer? So far, every layer has used the same activation, ReLU. We often want to use other types of activation however, so we need to update our code to support multiple types of activation functions. Make sure that you have completed every previous exercise before trying this one. a) Complete the b) Make a list with three activation functions(don’t call them yet! you can make a list with function names as elements, and then call these elements of the list later), two ReLU and one sigmoid. (If you add other functions than the ones defined at the start of the notebook, make sure everything is defined using autograd’s numpy wrapper, like above, since we want to use automatic differentiation on all of these functions later.) Then evaluate a network with three layers and these activation functions. So far, the feed forward function has taken one input vector as an input. This vector then undergoes a linear transformation and then an element-wise non-linear operation for each layer. This approach of sending one vector in at a time is great for interpreting how the network transforms data with its linear and non-linear operations, but not the best for numerical efficiency. Now, we want to be able to send many inputs through the network at once. This will make the code a bit harder to understand, but it will make it faster, and more compact. It will be worth the trouble. To process multiple inputs at once, while still performing the same operations, you will only need to flip a couple things around. a) Complete the function b) Update the b) Make a matrix of inputs with the shape (number of features, number of inputs), you choose the number of inputs and features per input. Then complete the function c) Create and evaluate a neural network with 4 inputs and layers with output sizes 12, 10, 3 and activations ReLU, ReLU, softmax. The final exercise will hopefully be very simple if everything has worked so far. You will evaluate your neural network on the iris data set (https://scikit-learn.org/1.5/auto_examples/datasets/plot_iris_dataset.html). This dataset contains data on 150 flowers of 3 different types which can be separated pretty well using the four features given for each flower, which includes the width and length of their leaves. You are not expected to do any training of the network or actual classification, unless you feel like it, in that case you can do exercise 5. You should use this batched approach moving forward, as it will lead to much more compact code. However, remember that each input is still treated separately, and that you will need to keep in mind the transposed weight matrix and other details when implementing backpropagation. You will now evaluate your neural network on the iris data set (https://scikit-learn.org/1.5/auto_examples/datasets/plot_iris_dataset.html). This dataset contains data on 150 flowers of 3 different types which can be separated pretty well using the four features given for each flower, which includes the width and length of their leaves. You are will later train your network to actually make good predictions. c) Loop over the iris dataset( a) What should the input size for the network be with this dataset? What should the output shape of the last layer be? b) Create a network with two hidden layers, the first with sigmoid activation and the last with softmax, the first layer should have 8 “nodes”, the second has the number of nodes you found in exercise a). Softmax returns a “probability distribution”, in the sense that the numbers in the output are positive and add up to 1 and, their magnitude are in some sense relative to their magnitude before going through the softmax function. Remember to use the batched version of the create_layers and feed forward functions. c) Evaluate your model on the entire iris dataset! For later purposes, we will split the data into train and test sets, and compute gradients on smaller batches of the training data. But for now, evaluate the network on the whole thing at once. d) Compute the accuracy of your model using the accuracy function defined above. Recreate your model a couple times and see how the accuracy changes. a) Make the iris target values into one-hot vectors. b) Define the cross-entropy loss function to evaluate the performance of your network on the data set. c) Use the autograd package to take the gradient of the cross entropy wrt. the weights and biases of the network. d) Use gradient descent of some sort to optimize the parameters. e) Evaluate the accuracy of the network. e) Show off how you did in a group session! To be able to actually do anything useful with your neural network, you need to train it. For this, we need a cost function and a way to take the gradient of the cost function wrt. the network parameters. The following exercises guide you through taking the gradient using autograd, and updating the network parameters using the gradient. Feel free to implement gradient methods like ADAM if you finish everything. The cross-entropy loss function can evaluate performance on classification tasks. It sees if your prediction is “most certain” on the correct target. a) What shape should the gradient of the cost function wrt. weights and biases be? b) Use the c) Finish the e) What do we call the gradient method used above? d) Train your network and see how the accuracy changes! Make a plot if you want. e) How high of an accuracy is it possible to acheive with a neural network on this dataset, if we use the whole thing as training data?
@@ -3610,10 +3620,10 @@ features).-0.009134699065945493
-4.0244965108017645
-[[0.85613835 2.50655379]
- [2.50655379 8.3404509 ]]
+
0.04718566894028431
+4.11080997912276
+[[ 1.10517643 3.48455788]
+ [ 3.48455788 12.00216162]]
0.07971187802560528
-1.800782161095708
-[[1. 0.59411814]
- [0.59411814 1. ]]
+
0.07836997022107646
+1.1378267322316808
+[[1. 0.63980097]
+ [0.63980097 1. ]]
[[ 0.81395716 1.89155934]
- [-1.34726166 -4.13453411]
- [-0.46229544 -2.34061974]
- [ 0.24429334 1.4051634 ]
- [ 0.41971814 1.6405671 ]
- [ 2.02456235 5.03973227]
- [-1.97311824 -4.72521196]
- [ 0.10738656 0.24578123]
- [-0.52702419 -2.34023682]
- [ 0.69978197 3.31779928]]
+
[[ 1.34931214 3.06139439]
+ [-0.44476964 -2.60794187]
+ [ 0.02225493 0.16388664]
+ [-1.91193672 -3.82324216]
+ [-0.2044881 -1.56027537]
+ [-1.15572395 -3.25982474]
+ [ 0.94217756 1.49888671]
+ [ 0.28472162 2.92474572]
+ [ 2.38943 7.14118216]
+ [-1.27097785 -3.5388115 ]]
0 1
-0 0.813957 1.891559
-1 -1.347262 -4.134534
-2 -0.462295 -2.340620
-3 0.244293 1.405163
-4 0.419718 1.640567
-5 2.024562 5.039732
-6 -1.973118 -4.725212
-7 0.107387 0.245781
-8 -0.527024 -2.340237
-9 0.699782 3.317799
+0 1.349312 3.061394
+1 -0.444770 -2.607942
+2 0.022255 0.163887
+3 -1.911937 -3.823242
+4 -0.204488 -1.560275
+5 -1.155724 -3.259825
+6 0.942178 1.498887
+7 0.284722 2.924746
+8 2.389430 7.141182
+9 -1.270978 -3.538811
0 1
-0 1.000000 0.969413
-1 0.969413 1.000000
+0 1.000000 0.950873
+1 0.950873 1.000000
0 1 2 3 4 5 6 7 \
0 0.0 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000
-1 0.0 0.092746 0.090302 0.090561 0.088058 0.085463 0.081530 0.078898
-2 0.0 0.090302 0.088694 0.089052 0.086797 0.084423 0.080286 0.077782
-3 0.0 0.090561 0.089052 0.095106 0.092533 0.089858 0.089404 0.086489
-4 0.0 0.088058 0.086797 0.092533 0.090114 0.087585 0.086913 0.084127
-5 0.0 0.085463 0.084423 0.089858 0.087585 0.085197 0.084340 0.081681
-6 0.0 0.081530 0.080286 0.089404 0.086913 0.084340 0.086471 0.083596
-7 0.0 0.078898 0.077782 0.086489 0.084127 0.081681 0.083596 0.080849
-8 0.0 0.076334 0.075334 0.083645 0.081405 0.079080 0.080793 0.078170
-9 0.0 0.073841 0.072946 0.080875 0.078753 0.076543 0.078067 0.075562
-10 0.0 0.072973 0.071789 0.082361 0.079982 0.077541 0.081296 0.078544
-11 0.0 0.070485 0.069391 0.079518 0.077254 0.074928 0.078454 0.075823
-12 0.0 0.068088 0.067077 0.076777 0.074622 0.072404 0.075712 0.073198
-13 0.0 0.065778 0.064845 0.074134 0.072083 0.069970 0.073071 0.070667
-14 0.0 0.063554 0.062693 0.071588 0.069637 0.067623 0.070527 0.068229
+1 0.0 0.074334 0.080585 0.077061 0.078751 0.080220 0.070657 0.071406
+2 0.0 0.080585 0.088425 0.082009 0.084289 0.086338 0.074009 0.075052
+3 0.0 0.077061 0.082009 0.085147 0.086339 0.087297 0.081324 0.081796
+4 0.0 0.078751 0.084289 0.086339 0.087789 0.089007 0.081926 0.082537
+5 0.0 0.080220 0.086338 0.087297 0.089007 0.090492 0.082307 0.083061
+6 0.0 0.070657 0.074009 0.081324 0.081926 0.082307 0.079874 0.080032
+7 0.0 0.071406 0.075052 0.081796 0.082537 0.083061 0.080032 0.080271
+8 0.0 0.072148 0.076101 0.082240 0.083128 0.083801 0.080150 0.080474
+9 0.0 0.072902 0.077180 0.082670 0.083714 0.084548 0.080237 0.080651
+10 0.0 0.063646 0.065859 0.075320 0.075498 0.075472 0.075478 0.075409
+11 0.0 0.064071 0.066452 0.075576 0.075838 0.075897 0.075542 0.075525
+12 0.0 0.064514 0.067074 0.075838 0.076189 0.076340 0.075602 0.075639
+13 0.0 0.064980 0.067731 0.076108 0.076555 0.076803 0.075658 0.075753
+14 0.0 0.065472 0.068429 0.076389 0.076938 0.077292 0.075711 0.075868
8 9 10 11 12 13 14
0 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000
-1 0.076334 0.073841 0.072973 0.070485 0.068088 0.065778 0.063554
-2 0.075334 0.072946 0.071789 0.069391 0.067077 0.064845 0.062693
-3 0.083645 0.080875 0.082361 0.079518 0.076777 0.074134 0.071588
-4 0.081405 0.078753 0.079982 0.077254 0.074622 0.072083 0.069637
-5 0.079080 0.076543 0.077541 0.074928 0.072404 0.069970 0.067623
-6 0.080793 0.078067 0.081296 0.078454 0.075712 0.073071 0.070527
-7 0.078170 0.075562 0.078544 0.075823 0.073198 0.070667 0.068229
-8 0.075609 0.073115 0.075864 0.073260 0.070747 0.068324 0.065989
-9 0.073115 0.070730 0.073260 0.070769 0.068364 0.066044 0.063809
-10 0.075864 0.073260 0.077608 0.074866 0.072223 0.069676 0.067224
-11 0.073260 0.070769 0.074866 0.072241 0.069711 0.067273 0.064924
-12 0.070747 0.068364 0.072223 0.069711 0.067288 0.064954 0.062705
-13 0.068324 0.066044 0.069676 0.067273 0.064954 0.062719 0.060565
-14 0.065989 0.063809 0.067224 0.064924 0.062705 0.060565 0.058503
+1 0.072148 0.072902 0.063646 0.064071 0.064514 0.064980 0.065472
+2 0.076101 0.077180 0.065859 0.066452 0.067074 0.067731 0.068429
+3 0.082240 0.082670 0.075320 0.075576 0.075838 0.076108 0.076389
+4 0.083128 0.083714 0.075498 0.075838 0.076189 0.076555 0.076938
+5 0.083801 0.084548 0.075472 0.075897 0.076340 0.076803 0.077292
+6 0.080150 0.080237 0.075478 0.075542 0.075602 0.075658 0.075711
+7 0.080474 0.080651 0.075409 0.075525 0.075639 0.075753 0.075868
+8 0.080766 0.081038 0.075293 0.075463 0.075634 0.075809 0.075988
+9 0.081038 0.081411 0.075136 0.075363 0.075595 0.075834 0.076082
+10 0.075293 0.075136 0.072406 0.072329 0.072240 0.072140 0.072028
+11 0.075463 0.075363 0.072329 0.072286 0.072234 0.072173 0.072101
+12 0.075634 0.075595 0.072240 0.072234 0.072220 0.072199 0.072171
+13 0.075809 0.075834 0.072140 0.072173 0.072199 0.072221 0.072238
+14 0.075988 0.076082 0.072028 0.072101 0.072171 0.072238 0.072303
[2. 2.]
-Training MSE for OLS
+
Training MSE for OLS
3.0
+
Runtime: 0.146141 sec
+
Runtime: 0.154751 sec
Jackknife Statistics :
original bias std. error
- 100.139 100.129 0.148776
+ 99.9896 99.9796 0.149524
Bootstrap Statistics :
original bias std. error
- 99.989 15.1792 99.9878 0.152149
+ 100.307 14.9693 100.309 0.149416
Polynomial degree: 2
Error: 0.10398646080125035
Bias^2: 0.1007711427354898
Var: 0.0032153180657605116
@@ -1315,7 +1327,9 @@ Error: 0.037813671417389005
Bias^2: 0.033657685071527665
Var: 0.00415598634586135
0.037813671417389005 >= 0.033657685071527665 + 0.00415598634586135 = 0.03781367141738902
-Polynomial degree: 7
+
Polynomial degree: 7
Error: 0.02760977349102253
Bias^2: 0.022999498260366312
Var: 0.004610275230656212
@@ -1342,21 +1356,21 @@ Error: 0.07160048164233104
Bias^2: 0.014436800088904942
Var: 0.05716368155342608
0.07160048164233104 >= 0.014436800088904942 + 0.05716368155342608 = 0.07160048164233102
-Polynomial degree: 12
+
Polynomial degree: 12
Error: 0.11547777218872497
Bias^2: 0.01628578269596628
Var: 0.09919198949275869
0.11547777218872497 >= 0.01628578269596628 + 0.09919198949275869 = 0.11547777218872497
-
Polynomial degree: 13
+Polynomial degree: 13
Error: 0.22842468702219465
Bias^2: 0.01975416527185249
Var: 0.20867052175034223
0.22842468702219465 >= 0.01975416527185249 + 0.20867052175034223 = 0.2284246870221947
+
Degree of polynomial: 25
+
Degree of polynomial: 24
+Mean squared error on training data: 0.00084707
+Mean squared error on test data: 1325.26124692
+Degree of polynomial: 25
Mean squared error on training data: 0.00079125
Mean squared error on test data: 129012.83870189
Degree of polynomial: 26
@@ -1661,19 +1675,19 @@ Mean squared error on test data: 18388.59354079
Degree of polynomial: 27
Mean squared error on training data: 0.00069123
Mean squared error on test data: 2351.97979891
-Degree of polynomial: 28
-Mean squared error on training data: 0.00062592
-Mean squared error on test data: 3983.63037846
Degree of polynomial: 29
+
Degree of polynomial: 28
+Mean squared error on training data: 0.00062592
+Mean squared error on test data: 3983.63037846
+Degree of polynomial: 29
Mean squared error on training data: 0.00060704
Mean squared error on test data: 3262.26814548
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/626635268.py:73: RuntimeWarning: divide by zero encountered in log10
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/626635268.py:73: RuntimeWarning: divide by zero encountered in log10
plt.plot(polynomial, np.log10(trainingerror), label='Training Error')
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/626635268.py:74: RuntimeWarning: divide by zero encountered in log10
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/626635268.py:74: RuntimeWarning: divide by zero encountered in log10
plt.plot(polynomial, np.log10(testerror), label='Test Error')
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/3817475779.py:63: RuntimeWarning: divide by zero encountered in log10
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/3817475779.py:63: RuntimeWarning: divide by zero encountered in log10
plt.plot(polynomial, np.log10(estimated_mse_sklearn), label='Test Error')
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/4162706317.py:7: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/4162706317.py:7: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/3777801602.py:7: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/3777801602.py:7: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/438060758.py:10: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/438060758.py:10: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_22458/3544313922.py:9: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57183/3544313922.py:9: UserWarning: set_ticklabels() should only be used with a fixed number of ticks, i.e. after set_ticks() or using a FixedLocator.
cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
0%| | 0/10 [00:00<?, ?it/s]
+
0%| | 0/10 [00:00<?, ?it/s]
/Users/mhjensen/miniforge3/envs/myenv/lib/python3.9/site-packages/sklearn/linear_model/_coordinate_descent.py:628: ConvergenceWarning: Objective did not converge. You might want to increase the number of iterations, check the scale of the features or consider increasing regularisation. Duality gap: 3.924e+00, tolerance: 1.797e+00
model = cd_fast.enet_coordinate_descent(
- 10%|███████████████████▏ | 1/10 [00:00<00:06, 1.29it/s]
+ 10%|█████████████▍ | 1/10 [00:00<00:07, 1.14it/s]
20%|██████████████████████████████████████▍ | 2/10 [00:01<00:06, 1.18it/s]
+
20%|██████████████████████████▊ | 2/10 [00:01<00:06, 1.16it/s]
30%|█████████████████████████████████████████████████████████▌ | 3/10 [00:02<00:05, 1.29it/s]
+
30%|████████████████████████████████████████▏ | 3/10 [00:02<00:05, 1.31it/s]
40%|████████████████████████████████████████████████████████████████████████████▊ | 4/10 [00:03<00:04, 1.32it/s]
+
40%|█████████████████████████████████████████████████████▌ | 4/10 [00:03<00:04, 1.39it/s]
50%|████████████████████████████████████████████████████████████████████████████████████████████████ | 5/10 [00:03<00:03, 1.54it/s]
+
50%|███████████████████████████████████████████████████████████████████ | 5/10 [00:03<00:03, 1.38it/s]
60%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████▏ | 6/10 [00:04<00:02, 1.65it/s]
+
60%|████████████████████████████████████████████████████████████████████████████████▍ | 6/10 [00:04<00:02, 1.44it/s]
70%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████▍ | 7/10 [00:04<00:02, 1.44it/s]
+
70%|█████████████████████████████████████████████████████████████████████████████████████████████▊ | 7/10 [00:05<00:02, 1.45it/s]
80%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████▌ | 8/10 [00:05<00:01, 1.50it/s]
+
80%|███████████████████████████████████████████████████████████████████████████████████████████████████████████▏ | 8/10 [00:05<00:01, 1.55it/s]
90%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████▊ | 9/10 [00:06<00:00, 1.47it/s]
+
90%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████▌ | 9/10 [00:06<00:00, 1.54it/s]
100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 10/10 [00:06<00:00, 1.52it/s]
+
100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 10/10 [00:07<00:00, 1.48it/s]
100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 10/10 [00:06<00:00, 1.45it/s]
+
100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 10/10 [00:07<00:00, 1.43it/s]
diff --git a/doc/LectureNotes/_build/html/chapter4.html b/doc/LectureNotes/_build/html/chapter4.html
index 3e81bb26c..1abcd6de0 100644
--- a/doc/LectureNotes/_build/html/chapter4.html
+++ b/doc/LectureNotes/_build/html/chapter4.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+ 2nd degree coefficients:
-zero power: -5.030164788184997
-first power: 0.001268544905013411
-second power: -5.234332597247826e-05
+zero power: -0.22158725223474995
+first power: 0.24121476598003452
+second power: -0.0009532583857747255
diff --git a/doc/LectureNotes/_build/html/chapter7.html b/doc/LectureNotes/_build/html/chapter7.html
index 1e6480c23..2b7035e16 100644
--- a/doc/LectureNotes/_build/html/chapter7.html
+++ b/doc/LectureNotes/_build/html/chapter7.html
@@ -323,6 +323,16 @@ const thebe_selector_output = ".output, .cell_output"
Week 41 Neural networks and constructing a neural network code
+ -0.012423940191689783
-4.101008878523571
-[[0.89527291 2.65532045]
- [2.65532045 8.81987609]]
+
0.1001408041761458
+4.2807716628772665
+[[ 1.15654145 3.54867722]
+ [ 3.54867722 11.70485195]]
0.08705631913312815
-1.7026908764394864
-[[1. 0.65870313]
- [0.65870313 1. ]]
+
0.09543871010617433
+1.6888043337746685
+[[1. 0.7167077]
+ [0.7167077 1. ]]
[[-1.7755649 -4.56778296]
- [-0.81015037 -2.80072356]
- [ 0.73628249 1.95206335]
- [ 0.97366347 1.61130099]
- [ 0.7271324 1.97965627]
- [ 0.36881837 0.56037913]
- [-1.33163086 -2.59391196]
- [-0.68953877 -1.58298728]
- [ 0.19982428 -1.08010965]
- [ 1.60116388 6.52211567]]
+
[[-0.20575734 0.01384583]
+ [-0.89876098 -3.04065686]
+ [-0.76289128 -3.17080691]
+ [-0.0334136 0.16124569]
+ [ 2.73970542 9.28885103]
+ [ 0.75413023 2.98474769]
+ [-1.87894459 -5.48121459]
+ [-1.26814205 -2.4848097 ]
+ [ 0.18114057 -0.9889962 ]
+ [ 1.37293361 2.71779401]]
0 1
-0 -1.775565 -4.567783
-1 -0.810150 -2.800724
-2 0.736282 1.952063
-3 0.973663 1.611301
-4 0.727132 1.979656
-5 0.368818 0.560379
-6 -1.331631 -2.593912
-7 -0.689539 -1.582987
-8 0.199824 -1.080110
-9 1.601164 6.522116
- 0 1
-0 1.00000 0.94335
-1 0.94335 1.00000
+0 -0.205757 0.013846
+1 -0.898761 -3.040657
+2 -0.762891 -3.170807
+3 -0.033414 0.161246
+4 2.739705 9.288851
+5 0.754130 2.984748
+6 -1.878945 -5.481215
+7 -1.268142 -2.484810
+8 0.181141 -0.988996
+9 1.372934 2.717794
+ 0 1
+0 1.000000 0.970965
+1 0.970965 1.000000
0 1 2 3 4 5 6 7 \
0 0.0 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000
-1 0.0 0.088650 0.084209 0.092689 0.091653 0.090476 0.085764 0.085378
-2 0.0 0.084209 0.080559 0.088599 0.087876 0.087020 0.082478 0.082249
-3 0.0 0.092689 0.088599 0.102209 0.101292 0.100209 0.097667 0.097380
-4 0.0 0.091653 0.087876 0.101292 0.100523 0.099588 0.097017 0.096811
-5 0.0 0.090476 0.087020 0.100209 0.099588 0.098803 0.096203 0.096078
-6 0.0 0.085764 0.082478 0.097667 0.097017 0.096203 0.095425 0.095278
-7 0.0 0.085378 0.082249 0.097380 0.096811 0.096078 0.095278 0.095178
-8 0.0 0.084976 0.082002 0.097060 0.096570 0.095915 0.095090 0.095037
-9 0.0 0.084550 0.081730 0.096700 0.096287 0.095710 0.094857 0.094849
-10 0.0 0.077672 0.075070 0.090429 0.090002 0.089421 0.089826 0.089786
-11 0.0 0.077490 0.074976 0.090319 0.089940 0.089408 0.089800 0.089790
-12 0.0 0.077310 0.074882 0.090204 0.089872 0.089386 0.089763 0.089783
-13 0.0 0.077131 0.074787 0.090081 0.089795 0.089354 0.089714 0.089763
-14 0.0 0.076951 0.074688 0.089949 0.089708 0.089311 0.089653 0.089730
+1 0.0 0.090565 0.089065 0.093226 0.091674 0.090154 0.086287 0.084847
+2 0.0 0.089065 0.087946 0.092421 0.091068 0.089735 0.085988 0.084674
+3 0.0 0.093226 0.092421 0.102147 0.100816 0.099494 0.098144 0.096731
+4 0.0 0.091674 0.091068 0.100816 0.099622 0.098431 0.097115 0.095803
+5 0.0 0.090154 0.089735 0.099494 0.098431 0.097365 0.096077 0.094862
+6 0.0 0.086287 0.085988 0.098144 0.097115 0.096077 0.096630 0.095395
+7 0.0 0.084847 0.084674 0.096731 0.095803 0.094862 0.095395 0.094243
+8 0.0 0.083459 0.083405 0.095358 0.094527 0.093680 0.094189 0.093115
+9 0.0 0.082121 0.082180 0.094027 0.093288 0.092530 0.093011 0.092013
+10 0.0 0.078708 0.078711 0.091732 0.090935 0.090118 0.091871 0.090804
+11 0.0 0.077431 0.077523 0.090387 0.089668 0.088926 0.090626 0.089626
+12 0.0 0.076203 0.076378 0.089086 0.088441 0.087772 0.089417 0.088481
+13 0.0 0.075021 0.075274 0.087828 0.087255 0.086655 0.088242 0.087369
+14 0.0 0.073883 0.074212 0.086611 0.086107 0.085573 0.087102 0.086289
8 9 10 11 12 13 14
0 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000
-1 0.084976 0.084550 0.077672 0.077490 0.077310 0.077131 0.076951
-2 0.082002 0.081730 0.075070 0.074976 0.074882 0.074787 0.074688
-3 0.097060 0.096700 0.090429 0.090319 0.090204 0.090081 0.089949
-4 0.096570 0.096287 0.090002 0.089940 0.089872 0.089795 0.089708
-5 0.095915 0.095710 0.089421 0.089408 0.089386 0.089354 0.089311
-6 0.095090 0.094857 0.089826 0.089800 0.089763 0.089714 0.089653
-7 0.095037 0.094849 0.089786 0.089790 0.089783 0.089763 0.089730
-8 0.094941 0.094797 0.089704 0.089738 0.089760 0.089769 0.089763
-9 0.094797 0.094697 0.089576 0.089639 0.089690 0.089726 0.089748
-10 0.089704 0.089576 0.085655 0.085690 0.085712 0.085722 0.085716
-11 0.089738 0.089639 0.085690 0.085747 0.085790 0.085819 0.085833
-12 0.089760 0.089690 0.085712 0.085790 0.085853 0.085902 0.085935
-13 0.089769 0.089726 0.085722 0.085819 0.085902 0.085970 0.086021
-14 0.089763 0.089748 0.085716 0.085833 0.085935 0.086021 0.086092
+1 0.083459 0.082121 0.078708 0.077431 0.076203 0.075021 0.073883
+2 0.083405 0.082180 0.078711 0.077523 0.076378 0.075274 0.074212
+3 0.095358 0.094027 0.091732 0.090387 0.089086 0.087828 0.086611
+4 0.094527 0.093288 0.090935 0.089668 0.088441 0.087255 0.086107
+5 0.093680 0.092530 0.090118 0.088926 0.087772 0.086655 0.085573
+6 0.094189 0.093011 0.091871 0.090626 0.089417 0.088242 0.087102
+7 0.093115 0.092013 0.090804 0.089626 0.088481 0.087369 0.086289
+8 0.092064 0.091034 0.089755 0.088642 0.087560 0.086508 0.085486
+9 0.091034 0.090075 0.088726 0.087675 0.086653 0.085659 0.084694
+10 0.089755 0.088726 0.088455 0.087327 0.086227 0.085155 0.084112
+11 0.088642 0.087675 0.087327 0.086256 0.085212 0.084195 0.083203
+12 0.087560 0.086653 0.086227 0.085212 0.084222 0.083257 0.082316
+13 0.086508 0.085659 0.085155 0.084195 0.083257 0.082342 0.081450
+14 0.085486 0.084694 0.084112 0.083203 0.082316 0.081450 0.080605
0 1
-0 3.914672 1.954823
-1 1.954823 1.963858
-[[3.91467223 1.95482298]
- [1.95482298 1.96385798]]
+0 4.114499 2.071143
+1 2.071143 2.061388
+
[[4.11449851 2.07114326]
+ [2.07114326 2.0613875 ]]
Centered covariance using own code
-[[3.91467223 1.95482298]
- [1.95482298 1.96385798]]
+[[4.11449851 2.07114326]
+ [2.07114326 2.0613875 ]]
@@ -1206,16 +1218,16 @@ questions.
Eigenvalues of Covariance matrix
-5.123928000581825
-0.7546022135629743
+5.399533503407795
+0.776352510835556
First eigenvector
-[0.85043503 0.5260801 ]
+[0.84973247 0.52721412]
Second eigenvector
-[-0.5260801 0.85043503]
+[-0.52721412 0.84973247]
Eigenvector of largest eigenvalue
-[0.85043503 0.5260801 ]
+[-0.84973247 -0.52721412]
Own inversion
-[[4.04881585]
- [2.96745096]]
-Eigenvalues of Hessian Matrix:[0.37175588 4.15690073]
+[[4.1729993]
+ [3.0170097]]
+Eigenvalues of Hessian Matrix:[0.28192769 4.68753434]
theta from own gd
-[[4.04881585]
- [2.96745096]]
+[[4.1729993]
+ [3.0170097]]
theta from own sdg
-[[4.04015455]
- [2.95745651]]
+[[4.14043884]
+ [2.99071523]]
@@ -934,14 +934,14 @@ first example shows results with ordinary leats squares.
Own inversion
-[[3.37007195]
- [3.48441798]]
-Eigenvalues of Hessian Matrix:[0.32411274 4.30450049]
+[[4.03696458]
+ [3.0324793 ]]
+Eigenvalues of Hessian Matrix:[0.30959659 4.4150026 ]
theta from own gd
-[[3.37007195]
- [3.48441798]]
+[[4.03696458]
+ [3.0324793 ]]
@@ -1012,73 +1012,73 @@ Eigenvalues of Hessian Matrix:[0.32411274 4.30450049]
Own inversion
[[4.]
[3.]]
-Eigenvalues of Hessian Matrix:[0.27227895 4.125757 ]
-0 [-11.75978215] [-12.89770892]
-1 [-0.06807123] [0.06136823]
-2 [-0.06357887] [0.05731824]
-3 [-0.05938299] [0.05353553]
-4 [-0.05546402] [0.05000245]
-5 [-0.05180367] [0.04670255]
-6 [-0.04838489] [0.04362042]
-7 [-0.04519174] [0.04074169]
-8 [-0.04220931] [0.03805295]
-9 [-0.03942371] [0.03554165]
-10 [-0.03682195] [0.03319608]
-11 [-0.03439189] [0.03100531]
-12 [-0.0321222] [0.02895911]
-13 [-0.0300023] [0.02704796]
-14 [-0.0280223] [0.02526293]
-15 [-0.02617297] [0.02359571]
-16 [-0.02444569] [0.02203851]
-17 [-0.0228324] [0.02058408]
-18 [-0.02132557] [0.01922564]
-19 [-0.01991819] [0.01795684]
-20 [-0.0186037] [0.01677178]
-21 [-0.01737595] [0.01566493]
-22 [-0.01622922] [0.01463112]
-23 [-0.01515818] [0.01366555]
-24 [-0.01415781] [0.01276369]
-25 [-0.01322347] [0.01192135]
-26 [-0.01235079] [0.0111346]
-27 [-0.0115357] [0.01039977]
-28 [-0.0107744] [0.00971344]
-29 [-0.01006335] [0.0090724]
+Eigenvalues of Hessian Matrix:[0.23469347 4.90407685]
+0 [-18.02987043] [-22.42501752]
+1 [-0.32329897] [0.25206371]
+2 [-0.30782691] [0.24000075]
+3 [-0.2930953] [0.22851508]
+4 [-0.27906869] [0.21757908]
+5 [-0.26571335] [0.20716644]
+6 [-0.25299716] [0.19725211]
+7 [-0.24088952] [0.18781225]
+8 [-0.22936132] [0.17882416]
+9 [-0.21838482] [0.17026621]
+10 [-0.20793362] [0.16211781]
+11 [-0.19798258] [0.15435937]
+12 [-0.18850776] [0.14697222]
+13 [-0.17948638] [0.1399386]
+14 [-0.17089674] [0.13324158]
+15 [-0.16271817] [0.12686507]
+16 [-0.15493099] [0.12079371]
+17 [-0.14751649] [0.11501291]
+18 [-0.14045682] [0.10950876]
+19 [-0.13373501] [0.10426802]
+20 [-0.12733488] [0.09927808]
+21 [-0.12124104] [0.09452695]
+22 [-0.11543883] [0.09000319]
+23 [-0.10991429] [0.08569593]
+24 [-0.10465415] [0.08159479]
+25 [-0.09964573] [0.07768993]
+26 [-0.094877] [0.07397193]
+27 [-0.09033649] [0.07043187]
+28 [-0.08601328] [0.06706123]
+29 [-0.08189696] [0.06385189]
theta from own gd
-[[3.96547946]
- [3.03112129]]
-0 [-0.00939922] [0.00847367]
-1 [-0.00877892] [0.00791445]
-2 [-0.00801346] [0.00722437]
-3 [-0.00725498] [0.00654058]
-4 [-0.00654864] [0.00590379]
-5 [-0.00590456] [0.00532314]
-6 [-0.00532167] [0.00479764]
-7 [-0.0047956] [0.00432337]
-8 [-0.00432129] [0.00389577]
-9 [-0.00389382] [0.00351039]
-10 [-0.0035086] [0.00316311]
-11 [-0.00316149] [0.00285017]
-12 [-0.00284871] [0.0025682]
-13 [-0.00256688] [0.00231412]
-14 [-0.00231293] [0.00208517]
-15 [-0.0020841] [0.00187888]
-16 [-0.00187791] [0.00169299]
-17 [-0.00169212] [0.0015255]
-18 [-0.00152471] [0.00137458]
-19 [-0.00137387] [0.00123858]
-20 [-0.00123795] [0.00111605]
-21 [-0.00111547] [0.00100563]
-22 [-0.00100511] [0.00090614]
-23 [-0.00090567] [0.00081649]
-24 [-0.00081607] [0.00073571]
-25 [-0.00073534] [0.00066293]
-26 [-0.00066259] [0.00059734]
-27 [-0.00059703] [0.00053824]
-28 [-0.00053797] [0.00048499]
-29 [-0.00048474] [0.00043701]
+[[3.66774691]
+ [3.25904489]]
+0 [-0.07797763] [0.06079614]
+1 [-0.07424587] [0.05788664]
+2 [-0.06957317] [0.05424351]
+3 [-0.06484181] [0.05055465]
+4 [-0.06031928] [0.04702861]
+5 [-0.05607583] [0.04372016]
+6 [-0.05211919] [0.04063532]
+7 [-0.04843794] [0.03776519]
+8 [-0.04501548] [0.03509683]
+9 [-0.04183444] [0.0326167]
+10 [-0.03887807] [0.03031173]
+11 [-0.03613058] [0.02816961]
+12 [-0.03357723] [0.02617887]
+13 [-0.03120433] [0.02432881]
+14 [-0.02899912] [0.02260949]
+15 [-0.02694975] [0.02101168]
+16 [-0.02504521] [0.01952679]
+17 [-0.02327527] [0.01814683]
+18 [-0.0216304] [0.01686439]
+19 [-0.02010178] [0.01567258]
+20 [-0.01868119] [0.014565]
+21 [-0.01736099] [0.01353569]
+22 [-0.01613409] [0.01257912]
+23 [-0.01499389] [0.01169016]
+24 [-0.01393427] [0.01086401]
+25 [-0.01294954] [0.01009625]
+26 [-0.01203439] [0.00938275]
+27 [-0.01118392] [0.00871967]
+28 [-0.01039355] [0.00810345]
+29 [-0.00965904] [0.00753078]
theta from own gd wth momentum
-[[3.99839581]
- [3.00144622]]
+[[3.96175251]
+ [3.02982009]]
Own inversion
-[[3.81818338]
- [3.12840176]]
-Eigenvalues of Hessian Matrix:[0.26024513 4.64291046]
-0 [-16.3824865] [-20.20867716]
-1 [9.71640303e-15] [1.05750493e-14]
-2 [-2.05998413e-17] [-2.11570114e-16]
-3 [6.96057795e-17] [1.90885733e-16]
-4 [6.96057795e-17] [1.90885733e-16]
+[[4.15451852]
+ [2.83230774]]
+Eigenvalues of Hessian Matrix:[0.30616802 4.24299211]
+0 [-10.57502449] [-11.57610367]
+1 [-4.47905601e-15] [-4.07372439e-16]
+2 [-6.9388939e-16] [-7.21432413e-16]
+3 [-6.9388939e-16] [-7.21432413e-16]
+4 [-6.9388939e-16] [-7.21432413e-16]
beta from own Newton code
-[[3.81818338]
- [3.12840176]]
+[[4.15451852]
+ [2.83230774]]
Own inversion
-[[3.6955259]
- [3.2809076]]
-Eigenvalues of Hessian Matrix:[0.31381731 4.50516278]
-theta from own gd
-[[3.6955259]
- [3.2809076]]
+[[4.04601419]
+ [3.12204312]]
+Eigenvalues of Hessian Matrix:[0.33604208 4.45709724]
+theta from own gd
+[[4.04601419]
+ [3.12204312]]
+
theta from own sdg
-[[3.6805151 ]
- [3.33045013]]
+[[4.02781444]
+ [3.13976073]]
Own inversion
-[[3.98751068]
- [2.92901115]]
-Eigenvalues of Hessian Matrix:[0.28244905 4.61501312]
+[[3.96417888]
+ [3.06634473]]
+Eigenvalues of Hessian Matrix:[0.32962444 4.18715465]
theta from own gd
-[[3.98556236]
- [2.93059014]]
+[[3.9639885]
+ [3.0665111]]
theta from own sdg with momentum
-[[4.01133846]
- [2.92452609]]
+[[4.00842216]
+ [3.14285244]]
theta from own AdaGrad
-[[1.99974365]
- [3.0013839 ]
- [3.99861193]]
+[[1.99969895]
+ [3.00167058]
+ [3.99835872]]
theta from own RMSprop
-[[1.99757634]
- [2.9983289 ]
- [3.99759503]]
+[[1.99852187]
+ [3.03868311]
+ [3.95744254]]
theta from own ADAM
-[[1.99993596]
- [3.00035483]
- [3.99963172]]
+[[1.99996471]
+ [3.00026784]
+ [3.99973141]]
[<matplotlib.lines.Line2D at 0x1360ef5e0>]
+
[<matplotlib.lines.Line2D at 0x11892e7f0>]
@@ -1688,7 +1690,7 @@ It provides composable transformations of Python+NumPy programs: differentiate,
<matplotlib.collections.PathCollection at 0x135f8b1f0>
+
<matplotlib.collections.PathCollection at 0x118995f70>
diff --git a/doc/LectureNotes/_build/html/exercisesweek42.html b/doc/LectureNotes/_build/html/exercisesweek42.html
index f044e2489..f4d0fc025 100644
--- a/doc/LectureNotes/_build/html/exercisesweek42.html
+++ b/doc/LectureNotes/_build/html/exercisesweek42.html
@@ -445,13 +445,23 @@ const thebe_selector_output = ".output, .cell_output"
Exercises week 42¶
-Overarching aims of the exercises this week¶
import autograd.numpy as np
-from autograd import grad
+
import autograd.numpy as np # We need to use this numpy wrapper to make automatic differentiation work later
+from sklearn import datasets
+import matplotlib.pyplot as plt
+from sklearn.metrics import accuracy_score
+
+
+# Defining some activation functions
+def ReLU(z):
+ return np.where(z > 0, z, 0)
+
+
+def sigmoid(z):
+ return 1 / (1 + np.exp(-z))
+
+
+def softmax(z):
+ """Compute softmax values for each set of scores in the rows of the matrix z.
+ Used with batched input data."""
+ e_z = np.exp(z - np.max(z, axis=0))
+ return e_z / np.sum(e_z, axis=1)[:, np.newaxis]
+
+
+def softmax_vec(z):
+ """Compute softmax values for each set of scores in the vector z.
+ Use this function when you use the activation function on one vector at a time"""
+ e_z = np.exp(z - np.max(z))
+ return e_z / np.sum(e_z)
Exercise 1¶
-np.random.seed(2024)
-
-def ReLU(z):
- return np.where(z > 0, z, 0)
-
-
-x = np.random.randn(2) # network input
+x = np.random.randn(2) # network input. This is a single input with two features
W1 = np.random.randn(4, 2) # first layer weights
b1with the correct shapeb1with the correct shape. (Run the next cell right after the previous to get the random generated values to line up with the test solution below)b1 = np.random.randn(4)
+
b1 = ...
z1 for the first layerz1 for the first layerz1 = W1 @ x + b1
+
z1 = ...
a1 for the first layer using the ReLU activation function defined earlier.a1 for the first layer using the ReLU activation function defined earlier.a1 = ReLU(z1)
+
a1 = ...
b1 with the randn function right after you define W1.sol1 = np.array([0.60610368, 4.0076268, 0.0, 0.56469864])
@@ -591,7 +633,41 @@ doconce format html exercisesweek41.do.txt -->
True
+
---------------------------------------------------------------------------
+TypeError Traceback (most recent call last)
+Cell In[6], line 3
+ 1 sol1 = np.array([0.60610368, 4.0076268, 0.0, 0.56469864])
+----> 3 print(np.allclose(a1, sol1))
+
+File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/autograd/tracer.py:48, in primitive.<locals>.f_wrapped(*args, **kwargs)
+ 46 return new_box(ans, trace, node)
+ 47 else:
+---> 48 return f_raw(*args, **kwargs)
+
+File <__array_function__ internals>:180, in allclose(*args, **kwargs)
+
+File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/numpy/core/numeric.py:2251, in allclose(a, b, rtol, atol, equal_nan)
+ 2180 @array_function_dispatch(_allclose_dispatcher)
+ 2181 def allclose(a, b, rtol=1.e-5, atol=1.e-8, equal_nan=False):
+ 2182 """
+ 2183 Returns True if two arrays are element-wise equal within a tolerance.
+ 2184
+ (...)
+ 2249
+ 2250 """
+-> 2251 res = all(isclose(a, b, rtol=rtol, atol=atol, equal_nan=equal_nan))
+ 2252 return bool(res)
+
+File <__array_function__ internals>:180, in isclose(*args, **kwargs)
+
+File ~/miniforge3/envs/myenv/lib/python3.9/site-packages/numpy/core/numeric.py:2358, in isclose(a, b, rtol, atol, equal_nan)
+ 2355 dt = multiarray.result_type(y, 1.)
+ 2356 y = asanyarray(y, dtype=dt)
+-> 2358 xfin = isfinite(x)
+ 2359 yfin = isfinite(y)
+ 2360 if all(xfin) and all(yfin):
+
+TypeError: ufunc 'isfinite' not supported for the input types, and the inputs could not be safely coerced to any supported types according to the casting rule ''safe''
Exercise 2¶
-W2 = np.random.randn(8, 4)
-b2 = np.random.randn(8)
+
W2 = ...
+b2 = ...
z2 and activation a2 for the second layer.z2 and activation a2 for the second layer.z2 = W2 @ a1
-a2 = ReLU(z2)
+
z2 = ...
+a2 = ...
print(a2.shape == (8,))
-
True
+
print(
+ np.allclose(np.exp(len(a2)), 2980.9579870417283)
+) # This should evaluate to True if a2 has the correct shape :)
layers of weight and bias tuples (W, b) for each layer, in order, with the correct shapes that we can use later as our network parameters.def create_layers(network_input_size, output_sizes):
+
def create_layers(network_input_size, layer_output_sizes):
layers = []
i_size = network_input_size
- for output_size in output_sizes:
- W = np.random.rand(output_size, i_size)
- b = np.random.rand(output_size)
+ for layer_output_size in layer_output_sizes:
+ W = ...
+ b = ...
layers.append((W, b))
- i_size = output_size
+ i_size = layer_output_size
return layers
z and activation a for each layer, and returns the final activation a. This is the complete feed-forward pass, a full neural network!z and activation a for each layer, with ReLU actication, and returns the final activation a. This is the complete feed-forward pass, a full neural network!def feed_forward(layers, input):
+
def feed_forward_all_relu(layers, input):
a = input
for W, b in layers:
- z = W @ a + b
- a = ReLU(z)
+ z = ...
+ a = ...
return a
Exercise 4¶
-create_layers function also accept a list of activation functions, which is used to add activation functions to each of the tuples in layers. Make new functions to not mess with the old ones.def create_layers_4(network_input_size, output_sizes, activation_funcs):
+
input_size = ...
+layer_output_sizes = [...]
+
+x = np.random.rand(input_size)
+layers = ...
+predict = ...
+print(predict)
+
Exercise 4 - Custom activation for each layer¶
+feed_forward function which accepts a list of activation functions as an argument, and which evaluates these activation functions at each layer.def feed_forward(input, layers, activations):
+ a = input
+ for (W, b), activation in zip(layers, activations):
+ z = ...
+ a = ...
+ return a
+
network_input_size = ...
+layer_output_sizes = [...]
+activations = [...]
+layers = ...
+
+x = np.random.randn(network_input_size)
+feed_forward(x, layers, activations)
+
Exercise 5 - Processing multiple inputs at once¶
+create_layers_batch so that the weight matrix is the transpose of what it was when you only sent in one input at a time.def create_layers_batch(network_input_size, layer_output_sizes):
layers = []
i_size = network_input_size
- for output_size, activation in zip(output_sizes, activation_funcs):
- W = np.random.rand(output_size, i_size)
- b = np.random.rand(output_size)
- layers.append((W, b, activation))
+ for layer_output_size in layer_output_sizes:
+ W = ...
+ b = ...
+ layers.append((W, b))
- i_size = output_size
+ i_size = layer_output_size
return layers
feed_forward function to support this change.feed_forward_batch so that you can process this matrix of inputs with only one matrix multiplication and one broadcasted vector addition per layer. (Hint: You will only need to swap two variable around from your previous implementation, but remember to test that you get the same results for equivelent inputs!)def feed_forward_4(layers, input):
- a = input
- for W, b, activation in layers:
- z = W @ a + b
- a = activation(z)
+
inputs = np.random.rand(1000, 4)
+
+
+def feed_forward_batch(inputs, layers, activations):
+ a = inputs
+ for (W, b), activation in zip(layers, activations):
+ z = ...
+ a = ...
return a
from scipy.special import softmax
-
-network_input_size = 4
-output_sizes = [12, 10, 3]
-activation_funcs = [ReLU, ReLU, softmax]
-layers = create_layers_4(network_input_size, output_sizes, activation_funcs)
+
network_input_size = ...
+layer_output_sizes = [...]
+activations = [...]
+layers = create_layers_batch(network_input_size, layer_output_sizes)
x = np.random.randn(network_input_size)
-predict = feed_forward_4(layers, x)
+feed_forward_batch(inputs, layers, activations)
Exercise 6 - Predicting on real data¶
+# Loading and plotting iris dataset
-from sklearn import datasets
-import matplotlib.pyplot as plt
-
-iris = datasets.load_iris()
+
iris = datasets.load_iris()
_, ax = plt.subplots()
scatter = ax.scatter(iris.data[:, 0], iris.data[:, 1], c=iris.target)
@@ -737,29 +859,114 @@ doconce format html exercisesweek41.do.txt -->
iris.data) and evaluate the network for each data point.# No need to change this cell! Just make sure it works!
-for x in iris.data:
- prediction = feed_forward_4(layers, x)
+
inputs = iris.data
+
+# Since each prediction is a vector with a score for each of the three types of flowers,
+# we need to make each target a vector with a 1 for the correct flower and a 0 for the others.
+targets = np.zeros((len(iris.data), 3))
+for i, t in enumerate(iris.target):
+ targets[i, t] = 1
+
+
+def accuracy(predictions, targets):
+ one_hot_predictions = np.zeros(predictions.shape)
+
+ for i, prediction in enumerate(predictions):
+ one_hot_predictions[i, np.argmax(prediction)] = 1
+ return accuracy_score(one_hot_predictions, targets)
+
...
+layers = ...
+
predictions = feed_forward_batch(inputs, layers, activations)
+
print(accuracy(predictions, targets))
Exercise 5 (Very optional and very hard :)¶
-Exercise 6 - Training on real data¶
+from autograd import grad
+
+
+def cost(input, layers, activations, target):
+ predict = feed_forward_batch(input, layers, activations)
+ return cross_entropy(predict, target)
+
+
+def cross_entropy(predict, target):
+ return np.sum(-target * np.log(predict))
+
+
+gradient_func = grad(
+ cross_entropy, 1
+) # Taking the gradient wrt. the second input to the cost function
+
gradient_func function to take the gradient of the cross entropy wrt. the weights and biases of the network. Check the shapes of what’s inside. What does the grad func from autograd actually do?layers_grad = gradient_func(inputs, layers, activations, targets) # Don't change this
+
train_network function.def train_network(
+ inputs, layers, activations, targets, learning_rate=0.001, epochs=100
+):
+ for i in range(epochs):
+ layers_grad = gradient_func(inputs, layers, activations, targets)
+ for (W, b), (W_g, b_g) in zip(layers, layers_grad):
+ W -= ...
+ b -= ...
+
...
+Old accuracy on training data: 0.1440501043841336
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:43: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:43: RuntimeWarning: overflow encountered in exp
exp_term = np.exp(self.z_o)
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/1630775253.py:44: RuntimeWarning: invalid value encountered in true_divide
self.probabilities = exp_term / np.sum(exp_term, axis=1, keepdims=True)
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+
/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
-/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_55841/953065564.py:4: RuntimeWarning: overflow encountered in exp
+/var/folders/td/3yk470mj5p931p9dtkk0y6jw0000gn/T/ipykernel_57345/953065564.py:4: RuntimeWarning: overflow encountered in exp
return 1/(1 + np.exp(-x))
Learning rate = 1.0
-Lambda = 0.1
-Accuracy score on test set: 0.08333333333333333
Learning rate = 1.0
-Lambda = 1.0
-Accuracy score on test set: 0.08888888888888889
+Lambda = 0.1
+Accuracy score on test set: 0.08333333333333333
Learning rate = 1.0
+Lambda = 1.0
+Accuracy score on test set: 0.08888888888888889
+
+Learning rate = 1.0
Lambda = 10.0
Accuracy score on test set: 0.09444444444444444
@@ -4355,9 +4354,8 @@ Accuracy score on test set: 0.10555555555555556
Learning rate = 10.0
Lambda = 0.01
Accuracy score on test set: 0.1388888888888889
-
Learning rate = 10.0
+
+Learning rate = 10.0
Lambda = 0.1
Accuracy score on test set: 0.11388888888888889