diff --git a/doc/pub/week44/html/._week44-bs033.html b/doc/pub/week44/html/._week44-bs033.html index 0c9950d88..499a798fb 100644 --- a/doc/pub/week44/html/._week44-bs033.html +++ b/doc/pub/week44/html/._week44-bs033.html @@ -349,20 +349,19 @@ infile = open# Read the experimental data with Pandas from IPython.display import display -grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade')) +grades = pd.read_csv(infile) grades = pd.DataFrame(grades) - +display(grades) # Features and targets X = grades.loc[:, grades.columns != 'Grade'].values y = grades.loc[:, grades.columns == 'Grade'].values - +print(X) # Create the encoder. encoder = OneHotEncoder(handle_unknown="ignore") # Assume for simplicity all features are categorical. encoder.fit(X) # Apply the encoder. X = encoder.transform(X) -print(X) # Then do a Classification tree tree_clf = DecisionTreeClassifier(max_depth=2) tree_clf.fit(X, y) @@ -376,6 +375,10 @@ export_graphviz( ) cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png' os.system(cmd) + + +#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] diff --git a/doc/pub/week44/html/week44-reveal.html b/doc/pub/week44/html/week44-reveal.html index 5a9da24aa..bb43e0325 100644 --- a/doc/pub/week44/html/week44-reveal.html +++ b/doc/pub/week44/html/week44-reveal.html @@ -1127,20 +1127,19 @@ infile = open(data_path(# Read the experimental data with Pandas from IPython.display import display -grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade')) +grades = pd.read_csv(infile) grades = pd.DataFrame(grades) - +display(grades) # Features and targets X = grades.loc[:, grades.columns != 'Grade'].values y = grades.loc[:, grades.columns == 'Grade'].values - +print(X) # Create the encoder. encoder = OneHotEncoder(handle_unknown="ignore") # Assume for simplicity all features are categorical. encoder.fit(X) # Apply the encoder. X = encoder.transform(X) -print(X) # Then do a Classification tree tree_clf = DecisionTreeClassifier(max_depth=2) tree_clf.fit(X, y) @@ -1154,6 +1153,10 @@ export_graphviz( ) cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png' os.system(cmd) + + +#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] diff --git a/doc/pub/week44/html/week44-solarized.html b/doc/pub/week44/html/week44-solarized.html index 287d68911..7635ffc80 100644 --- a/doc/pub/week44/html/week44-solarized.html +++ b/doc/pub/week44/html/week44-solarized.html @@ -1137,20 +1137,19 @@ infile = open(data_path(# Read the experimental data with Pandas from IPython.display import display -grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade')) +grades = pd.read_csv(infile) grades = pd.DataFrame(grades) - +display(grades) # Features and targets X = grades.loc[:, grades.columns != 'Grade'].values y = grades.loc[:, grades.columns == 'Grade'].values - +print(X) # Create the encoder. encoder = OneHotEncoder(handle_unknown="ignore") # Assume for simplicity all features are categorical. encoder.fit(X) # Apply the encoder. X = encoder.transform(X) -print(X) # Then do a Classification tree tree_clf = DecisionTreeClassifier(max_depth=2) tree_clf.fit(X, y) @@ -1164,6 +1163,10 @@ export_graphviz( ) cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png' os.system(cmd) + + +#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] diff --git a/doc/pub/week44/html/week44.html b/doc/pub/week44/html/week44.html index c4bc91cef..e7c71a386 100644 --- a/doc/pub/week44/html/week44.html +++ b/doc/pub/week44/html/week44.html @@ -1214,20 +1214,19 @@ infile = open# Read the experimental data with Pandas from IPython.display import display -grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade')) +grades = pd.read_csv(infile) grades = pd.DataFrame(grades) - +display(grades) # Features and targets X = grades.loc[:, grades.columns != 'Grade'].values y = grades.loc[:, grades.columns == 'Grade'].values - +print(X) # Create the encoder. encoder = OneHotEncoder(handle_unknown="ignore") # Assume for simplicity all features are categorical. encoder.fit(X) # Apply the encoder. X = encoder.transform(X) -print(X) # Then do a Classification tree tree_clf = DecisionTreeClassifier(max_depth=2) tree_clf.fit(X, y) @@ -1241,6 +1240,10 @@ export_graphviz( ) cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png' os.system(cmd) + + +#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] diff --git a/doc/pub/week44/ipynb/Datafiles/cancer.dot b/doc/pub/week44/ipynb/Datafiles/cancer.dot index 75af04abe..51d41287e 100644 --- a/doc/pub/week44/ipynb/Datafiles/cancer.dot +++ b/doc/pub/week44/ipynb/Datafiles/cancer.dot @@ -6,17 +6,17 @@ edge [fontname="helvetica"] ; 0 -> 1 [labeldistance=2.5, labelangle=45, headlabel="True"] ; 2 [label="worst concave points <= 0.135\ngini = 0.031\nsamples = 253\nvalue = [[249, 4]\n[4, 249]]", fillcolor="#e78946"] ; 1 -> 2 ; -3 [label="area error <= 48.975\ngini = 0.008\nsamples = 242\nvalue = [[241, 1]\n[1, 241]]", fillcolor="#e5833c"] ; +3 [label="radius error <= 0.643\ngini = 0.008\nsamples = 242\nvalue = [[241, 1]\n[1, 241]]", fillcolor="#e5833c"] ; 2 -> 3 ; 4 [label="gini = 0.0\nsamples = 239\nvalue = [[239, 0]\n[0, 239]]", fillcolor="#e58139"] ; 3 -> 4 ; -5 [label="perimeter error <= 4.249\ngini = 0.444\nsamples = 3\nvalue = [[2, 1]\n[1, 2]]", fillcolor="#fdf6f0"] ; +5 [label="worst compactness <= 0.085\ngini = 0.444\nsamples = 3\nvalue = [[2, 1]\n[1, 2]]", fillcolor="#fdf6f0"] ; 3 -> 5 ; 6 [label="gini = 0.0\nsamples = 1\nvalue = [[0, 1]\n[1, 0]]", fillcolor="#e58139"] ; 5 -> 6 ; 7 [label="gini = 0.0\nsamples = 2\nvalue = [[2, 0]\n[0, 2]]", fillcolor="#e58139"] ; 5 -> 7 ; -8 [label="mean texture <= 20.84\ngini = 0.397\nsamples = 11\nvalue = [[8, 3]\n[3, 8]]", fillcolor="#fae9dd"] ; +8 [label="worst texture <= 29.455\ngini = 0.397\nsamples = 11\nvalue = [[8, 3]\n[3, 8]]", fillcolor="#fae9dd"] ; 2 -> 8 ; 9 [label="gini = 0.0\nsamples = 8\nvalue = [[8, 0]\n[0, 8]]", fillcolor="#e58139"] ; 8 -> 9 ; @@ -34,7 +34,7 @@ edge [fontname="helvetica"] ; 14 -> 15 ; 16 [label="gini = 0.0\nsamples = 11\nvalue = [[11, 0]\n[0, 11]]", fillcolor="#e58139"] ; 15 -> 16 ; -17 [label="mean smoothness <= 0.086\ngini = 0.32\nsamples = 5\nvalue = [[1, 4]\n[4, 1]]", fillcolor="#f6d5bd"] ; +17 [label="smoothness error <= 0.005\ngini = 0.32\nsamples = 5\nvalue = [[1, 4]\n[4, 1]]", fillcolor="#f6d5bd"] ; 15 -> 17 ; 18 [label="gini = 0.0\nsamples = 1\nvalue = [[1, 0]\n[0, 1]]", fillcolor="#e58139"] ; 17 -> 18 ; diff --git a/doc/pub/week44/ipynb/Datafiles/cancer.png b/doc/pub/week44/ipynb/Datafiles/cancer.png index c0f1f7548..1edd37c90 100644 Binary files a/doc/pub/week44/ipynb/Datafiles/cancer.png and b/doc/pub/week44/ipynb/Datafiles/cancer.png differ diff --git a/doc/pub/week44/ipynb/Datafiles/ride.dot b/doc/pub/week44/ipynb/Datafiles/ride.dot index 4e3b65b18..2772fde19 100644 --- a/doc/pub/week44/ipynb/Datafiles/ride.dot +++ b/doc/pub/week44/ipynb/Datafiles/ride.dot @@ -1,7 +1,7 @@ digraph Tree { node [shape=box, style="filled, rounded", color="black", fontname="helvetica"] ; edge [fontname="helvetica"] ; -0 [label="X[9] <= 0.5\ngini = 0.48\nsamples = 15\nvalue = [4, 10, 1]", fillcolor="#93f1ba"] ; +0 [label="X[0] <= 0.5\ngini = 0.48\nsamples = 15\nvalue = [4, 10, 1]", fillcolor="#93f1ba"] ; 1 [label="X[1] <= 0.5\ngini = 0.408\nsamples = 14\nvalue = [4, 10, 0]", fillcolor="#88efb3"] ; 0 -> 1 [labeldistance=2.5, labelangle=45, headlabel="True"] ; 2 [label="gini = 0.48\nsamples = 10\nvalue = [4, 6, 0]", fillcolor="#bdf6d5"] ; diff --git a/doc/pub/week44/ipynb/ipynb-week44-src.tar.gz b/doc/pub/week44/ipynb/ipynb-week44-src.tar.gz index 5649744c5..4c52e88e6 100644 Binary files a/doc/pub/week44/ipynb/ipynb-week44-src.tar.gz and b/doc/pub/week44/ipynb/ipynb-week44-src.tar.gz differ diff --git a/doc/pub/week44/ipynb/week44.ipynb b/doc/pub/week44/ipynb/week44.ipynb index 7c8c7a8ee..8b5e8ac61 100644 --- a/doc/pub/week44/ipynb/week44.ipynb +++ b/doc/pub/week44/ipynb/week44.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "d98e2322", + "id": "303fdfce", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "9f95d00e", + "id": "89424210", "metadata": { "editable": true }, @@ -29,7 +29,7 @@ }, { "cell_type": "markdown", - "id": "1c3184d8", + "id": "6177c33f", "metadata": { "editable": true }, @@ -51,7 +51,7 @@ }, { "cell_type": "markdown", - "id": "03088d75", + "id": "07e99622", "metadata": { "editable": true }, @@ -69,7 +69,7 @@ }, { "cell_type": "markdown", - "id": "fb46f0fc", + "id": "cec56cc9", "metadata": { "editable": true }, @@ -100,7 +100,7 @@ }, { "cell_type": "markdown", - "id": "9acd9dda", + "id": "7d8a5e44", "metadata": { "editable": true }, @@ -120,7 +120,7 @@ }, { "cell_type": "markdown", - "id": "719a8716", + "id": "23a71044", "metadata": { "editable": true }, @@ -134,7 +134,7 @@ }, { "cell_type": "markdown", - "id": "6b490bb8", + "id": "f241ee74", "metadata": { "editable": true }, @@ -147,7 +147,7 @@ }, { "cell_type": "markdown", - "id": "4cb44801", + "id": "c428c3b2", "metadata": { "editable": true }, @@ -165,7 +165,7 @@ }, { "cell_type": "markdown", - "id": "c17529d5", + "id": "b340bd77", "metadata": { "editable": true }, @@ -188,7 +188,7 @@ }, { "cell_type": "markdown", - "id": "feb8f6fe", + "id": "94c90118", "metadata": { "editable": true }, @@ -211,7 +211,7 @@ }, { "cell_type": "markdown", - "id": "5890974d", + "id": "c8d71b3f", "metadata": { "editable": true }, @@ -222,7 +222,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "e08d6fa8", + "id": "287a7d14", "metadata": { "collapsed": false, "editable": true @@ -323,7 +323,7 @@ }, { "cell_type": "markdown", - "id": "d1daa9c8", + "id": "5f44ba61", "metadata": { "editable": true }, @@ -345,7 +345,7 @@ }, { "cell_type": "markdown", - "id": "92c73b5a", + "id": "76e979a6", "metadata": { "editable": true }, @@ -357,7 +357,7 @@ }, { "cell_type": "markdown", - "id": "1e70c569", + "id": "79c449fe", "metadata": { "editable": true }, @@ -368,7 +368,7 @@ }, { "cell_type": "markdown", - "id": "176bf038", + "id": "ea2d207a", "metadata": { "editable": true }, @@ -390,7 +390,7 @@ }, { "cell_type": "markdown", - "id": "ad3b71cb", + "id": "721908c8", "metadata": { "editable": true }, @@ -403,7 +403,7 @@ }, { "cell_type": "markdown", - "id": "5e4c164d", + "id": "c47e1ef0", "metadata": { "editable": true }, @@ -415,7 +415,7 @@ }, { "cell_type": "markdown", - "id": "89190413", + "id": "15b9c732", "metadata": { "editable": true }, @@ -425,7 +425,7 @@ }, { "cell_type": "markdown", - "id": "8b31354c", + "id": "ef0dc96b", "metadata": { "editable": true }, @@ -437,7 +437,7 @@ }, { "cell_type": "markdown", - "id": "f242d32e", + "id": "52ddf624", "metadata": { "editable": true }, @@ -447,7 +447,7 @@ }, { "cell_type": "markdown", - "id": "aaa6c9de", + "id": "e87adb62", "metadata": { "editable": true }, @@ -459,7 +459,7 @@ }, { "cell_type": "markdown", - "id": "dbd78a2f", + "id": "e5288f89", "metadata": { "editable": true }, @@ -492,7 +492,7 @@ }, { "cell_type": "markdown", - "id": "b2cef5e6", + "id": "03cdba4b", "metadata": { "editable": true }, @@ -516,7 +516,7 @@ }, { "cell_type": "markdown", - "id": "bd447e07", + "id": "7f3cca97", "metadata": { "editable": true }, @@ -528,7 +528,7 @@ }, { "cell_type": "markdown", - "id": "7a6cc693", + "id": "81908085", "metadata": { "editable": true }, @@ -540,7 +540,7 @@ }, { "cell_type": "markdown", - "id": "5ca7a58b", + "id": "00b84717", "metadata": { "editable": true }, @@ -568,7 +568,7 @@ }, { "cell_type": "markdown", - "id": "dc2385e7", + "id": "9cb838a8", "metadata": { "editable": true }, @@ -594,7 +594,7 @@ }, { "cell_type": "markdown", - "id": "61fb7288", + "id": "c77af063", "metadata": { "editable": true }, @@ -617,7 +617,7 @@ }, { "cell_type": "markdown", - "id": "9a02c899", + "id": "ab9b5809", "metadata": { "editable": true }, @@ -644,7 +644,7 @@ }, { "cell_type": "markdown", - "id": "15a2080f", + "id": "237b53b5", "metadata": { "editable": true }, @@ -663,7 +663,7 @@ }, { "cell_type": "markdown", - "id": "35093975", + "id": "7f0692e9", "metadata": { "editable": true }, @@ -675,7 +675,7 @@ }, { "cell_type": "markdown", - "id": "eca91c0b", + "id": "96b4cc20", "metadata": { "editable": true }, @@ -688,7 +688,7 @@ }, { "cell_type": "markdown", - "id": "5efd9a2b", + "id": "3c47a735", "metadata": { "editable": true }, @@ -700,7 +700,7 @@ }, { "cell_type": "markdown", - "id": "772f8fad", + "id": "668bb7e6", "metadata": { "editable": true }, @@ -710,7 +710,7 @@ }, { "cell_type": "markdown", - "id": "015e0c30", + "id": "c5bd680f", "metadata": { "editable": true }, @@ -722,7 +722,7 @@ }, { "cell_type": "markdown", - "id": "e483d862", + "id": "b83d5433", "metadata": { "editable": true }, @@ -732,7 +732,7 @@ }, { "cell_type": "markdown", - "id": "76daf24d", + "id": "4f78a0fe", "metadata": { "editable": true }, @@ -744,7 +744,7 @@ }, { "cell_type": "markdown", - "id": "20ce5b46", + "id": "a440adc9", "metadata": { "editable": true }, @@ -755,7 +755,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "1eccbfd3", + "id": "6e4efeb9", "metadata": { "collapsed": false, "editable": true @@ -799,7 +799,7 @@ }, { "cell_type": "markdown", - "id": "bcb30334", + "id": "0006059d", "metadata": { "editable": true }, @@ -810,7 +810,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "372ae619", + "id": "2600e2b7", "metadata": { "collapsed": false, "editable": true @@ -845,7 +845,7 @@ }, { "cell_type": "markdown", - "id": "754e81f0", + "id": "545efc0a", "metadata": { "editable": true }, @@ -858,7 +858,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "db2ea071", + "id": "a67593ad", "metadata": { "collapsed": false, "editable": true @@ -876,7 +876,7 @@ }, { "cell_type": "markdown", - "id": "35af5b50", + "id": "784a8c90", "metadata": { "editable": true }, @@ -890,7 +890,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "95a27cb4", + "id": "98309efc", "metadata": { "collapsed": false, "editable": true @@ -909,7 +909,7 @@ }, { "cell_type": "markdown", - "id": "fe5fc8fd", + "id": "800d4fd7", "metadata": { "editable": true }, @@ -929,7 +929,7 @@ }, { "cell_type": "markdown", - "id": "3c2850c2", + "id": "a783a947", "metadata": { "editable": true }, @@ -946,7 +946,7 @@ }, { "cell_type": "markdown", - "id": "f28005ca", + "id": "d9c975db", "metadata": { "editable": true }, @@ -958,7 +958,7 @@ }, { "cell_type": "markdown", - "id": "09b3ddeb", + "id": "56c0aa6d", "metadata": { "editable": true }, @@ -975,7 +975,7 @@ }, { "cell_type": "markdown", - "id": "58611816", + "id": "3f47228e", "metadata": { "editable": true }, @@ -988,7 +988,7 @@ }, { "cell_type": "markdown", - "id": "2d49e0d6", + "id": "9b713480", "metadata": { "editable": true }, @@ -1000,7 +1000,7 @@ }, { "cell_type": "markdown", - "id": "a0aba8b0", + "id": "9bda577d", "metadata": { "editable": true }, @@ -1010,7 +1010,7 @@ }, { "cell_type": "markdown", - "id": "3837f01d", + "id": "d2af50de", "metadata": { "editable": true }, @@ -1022,7 +1022,7 @@ }, { "cell_type": "markdown", - "id": "ffd106d5", + "id": "eff1087c", "metadata": { "editable": true }, @@ -1032,7 +1032,7 @@ }, { "cell_type": "markdown", - "id": "4e9a3380", + "id": "6a4fcf67", "metadata": { "editable": true }, @@ -1044,7 +1044,7 @@ }, { "cell_type": "markdown", - "id": "2c98346f", + "id": "0af59d3c", "metadata": { "editable": true }, @@ -1057,7 +1057,7 @@ }, { "cell_type": "markdown", - "id": "5af6dd83", + "id": "36d5b7e1", "metadata": { "editable": true }, @@ -1072,7 +1072,7 @@ }, { "cell_type": "markdown", - "id": "049c6c06", + "id": "ff15cbfd", "metadata": { "editable": true }, @@ -1098,7 +1098,7 @@ }, { "cell_type": "markdown", - "id": "d28ee51e", + "id": "6fda730a", "metadata": { "editable": true }, @@ -1126,7 +1126,7 @@ }, { "cell_type": "markdown", - "id": "9aedec9e", + "id": "91027fc4", "metadata": { "editable": true }, @@ -1143,7 +1143,7 @@ }, { "cell_type": "markdown", - "id": "249904b8", + "id": "d0e22658", "metadata": { "editable": true }, @@ -1157,7 +1157,7 @@ }, { "cell_type": "markdown", - "id": "ffc84821", + "id": "fbf239d1", "metadata": { "editable": true }, @@ -1173,7 +1173,7 @@ }, { "cell_type": "markdown", - "id": "aaef2ea7", + "id": "7ba39338", "metadata": { "editable": true }, @@ -1184,7 +1184,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "985f7f61", + "id": "6c1105f2", "metadata": { "collapsed": false, "editable": true @@ -1231,20 +1231,19 @@ "\n", "# Read the experimental data with Pandas\n", "from IPython.display import display\n", - "grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade'))\n", + "grades = pd.read_csv(infile)\n", "grades = pd.DataFrame(grades)\n", - "\n", + "display(grades)\n", "# Features and targets\n", "X = grades.loc[:, grades.columns != 'Grade'].values\n", "y = grades.loc[:, grades.columns == 'Grade'].values\n", - "\n", + "print(X)\n", "# Create the encoder.\n", "encoder = OneHotEncoder(handle_unknown=\"ignore\")\n", "# Assume for simplicity all features are categorical.\n", "encoder.fit(X) \n", "# Apply the encoder.\n", "X = encoder.transform(X)\n", - "print(X)\n", "# Then do a Classification tree\n", "tree_clf = DecisionTreeClassifier(max_depth=2)\n", "tree_clf.fit(X, y)\n", @@ -1257,12 +1256,16 @@ " filled=True\n", ")\n", "cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png'\n", - "os.system(cmd)" + "os.system(cmd)\n", + "\n", + "\n", + "#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam'])\n", + "#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']" ] }, { "cell_type": "markdown", - "id": "9255005b", + "id": "d902592f", "metadata": { "editable": true }, @@ -1306,7 +1309,7 @@ }, { "cell_type": "markdown", - "id": "3e22f450", + "id": "63daa840", "metadata": { "editable": true }, @@ -1317,7 +1320,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "a760bd54", + "id": "25b57fe3", "metadata": { "collapsed": false, "editable": true @@ -1395,7 +1398,7 @@ }, { "cell_type": "markdown", - "id": "7e0995e4", + "id": "c82a4c31", "metadata": { "editable": true }, @@ -1413,7 +1416,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "52b6994b", + "id": "66b02746", "metadata": { "collapsed": false, "editable": true @@ -1484,7 +1487,7 @@ }, { "cell_type": "markdown", - "id": "08d47f8e", + "id": "dde9cc81", "metadata": { "editable": true }, @@ -1495,7 +1498,7 @@ { "cell_type": "code", "execution_count": 9, - "id": "e7bb262f", + "id": "3891adc6", "metadata": { "collapsed": false, "editable": true @@ -1570,7 +1573,7 @@ }, { "cell_type": "markdown", - "id": "253c4fc0", + "id": "20e12c6d", "metadata": { "editable": true }, @@ -1581,7 +1584,7 @@ { "cell_type": "code", "execution_count": 10, - "id": "90b7f896", + "id": "937ee4a0", "metadata": { "collapsed": false, "editable": true @@ -1612,7 +1615,7 @@ }, { "cell_type": "markdown", - "id": "48e28e54", + "id": "05262e92", "metadata": { "editable": true }, @@ -1623,7 +1626,7 @@ { "cell_type": "code", "execution_count": 11, - "id": "1c4239fe", + "id": "2f5aa1d2", "metadata": { "collapsed": false, "editable": true @@ -1641,7 +1644,7 @@ { "cell_type": "code", "execution_count": 12, - "id": "31178820", + "id": "d0431be6", "metadata": { "collapsed": false, "editable": true @@ -1656,7 +1659,7 @@ }, { "cell_type": "markdown", - "id": "dd6a7eaa", + "id": "b7efc61a", "metadata": { "editable": true }, @@ -1667,7 +1670,7 @@ { "cell_type": "code", "execution_count": 13, - "id": "079592b6", + "id": "5c00cb35", "metadata": { "collapsed": false, "editable": true @@ -1717,7 +1720,7 @@ { "cell_type": "code", "execution_count": 14, - "id": "96eaa015", + "id": "03b39894", "metadata": { "collapsed": false, "editable": true @@ -1756,7 +1759,7 @@ }, { "cell_type": "markdown", - "id": "84fe3854", + "id": "d71f0ce9", "metadata": { "editable": true }, @@ -1780,7 +1783,7 @@ }, { "cell_type": "markdown", - "id": "187d746e", + "id": "4ad744d7", "metadata": { "editable": true }, @@ -1808,7 +1811,7 @@ }, { "cell_type": "markdown", - "id": "aeb79b8e", + "id": "919de051", "metadata": { "editable": true }, @@ -1839,7 +1842,7 @@ }, { "cell_type": "markdown", - "id": "928bbe7b", + "id": "b2e868f1", "metadata": { "editable": true }, @@ -1855,7 +1858,7 @@ }, { "cell_type": "markdown", - "id": "73027143", + "id": "e790cf95", "metadata": { "editable": true }, @@ -1879,7 +1882,7 @@ }, { "cell_type": "markdown", - "id": "18f0c929", + "id": "9cd3580f", "metadata": { "editable": true }, @@ -1910,7 +1913,7 @@ }, { "cell_type": "markdown", - "id": "ccef5549", + "id": "47969bcd", "metadata": { "editable": true }, @@ -1921,7 +1924,7 @@ { "cell_type": "code", "execution_count": 15, - "id": "e229b9ba", + "id": "67de73ef", "metadata": { "collapsed": false, "editable": true @@ -1969,7 +1972,7 @@ }, { "cell_type": "markdown", - "id": "b3767e7c", + "id": "606e8c8e", "metadata": { "editable": true }, @@ -1980,7 +1983,7 @@ { "cell_type": "code", "execution_count": 16, - "id": "afb5135b", + "id": "b107a1cf", "metadata": { "collapsed": false, "editable": true @@ -2014,7 +2017,7 @@ }, { "cell_type": "markdown", - "id": "a472c59e", + "id": "6abdb36a", "metadata": { "editable": true }, @@ -2027,7 +2030,7 @@ { "cell_type": "code", "execution_count": 17, - "id": "6c1dd58b", + "id": "00bce4fc", "metadata": { "collapsed": false, "editable": true @@ -2080,7 +2083,7 @@ }, { "cell_type": "markdown", - "id": "40f4a4b7", + "id": "dbbf0823", "metadata": { "editable": true }, @@ -2091,7 +2094,7 @@ { "cell_type": "code", "execution_count": 18, - "id": "c853270d", + "id": "deee652c", "metadata": { "collapsed": false, "editable": true @@ -2121,7 +2124,7 @@ { "cell_type": "code", "execution_count": 19, - "id": "0bbadee8", + "id": "4f831127", "metadata": { "collapsed": false, "editable": true @@ -2139,7 +2142,7 @@ { "cell_type": "code", "execution_count": 20, - "id": "64392209", + "id": "4f0ede67", "metadata": { "collapsed": false, "editable": true @@ -2159,7 +2162,7 @@ { "cell_type": "code", "execution_count": 21, - "id": "462868aa", + "id": "a6234319", "metadata": { "collapsed": false, "editable": true @@ -2176,7 +2179,7 @@ }, { "cell_type": "markdown", - "id": "c7cc4f29", + "id": "27964504", "metadata": { "editable": true }, @@ -2198,7 +2201,7 @@ }, { "cell_type": "markdown", - "id": "2b2ce4e3", + "id": "a014fe49", "metadata": { "editable": true }, @@ -2230,7 +2233,7 @@ }, { "cell_type": "markdown", - "id": "35e0799b", + "id": "44189d88", "metadata": { "editable": true }, @@ -2244,7 +2247,7 @@ { "cell_type": "code", "execution_count": 22, - "id": "08fb064e", + "id": "3d81803f", "metadata": { "collapsed": false, "editable": true @@ -2313,7 +2316,7 @@ }, { "cell_type": "markdown", - "id": "0d70c652", + "id": "1eabd234", "metadata": { "editable": true }, @@ -2336,7 +2339,7 @@ }, { "cell_type": "markdown", - "id": "c2b8b4e1", + "id": "d39aa2d2", "metadata": { "editable": true }, @@ -2348,7 +2351,7 @@ }, { "cell_type": "markdown", - "id": "88dd33a9", + "id": "c5d99442", "metadata": { "editable": true }, @@ -2373,7 +2376,7 @@ }, { "cell_type": "markdown", - "id": "bc8aefa6", + "id": "5e11503c", "metadata": { "editable": true }, @@ -2399,7 +2402,7 @@ }, { "cell_type": "markdown", - "id": "8d897dc6", + "id": "3af7a38e", "metadata": { "editable": true }, @@ -2410,7 +2413,7 @@ { "cell_type": "code", "execution_count": 23, - "id": "6fc7304c", + "id": "d8858b60", "metadata": { "collapsed": false, "editable": true @@ -2482,7 +2485,7 @@ }, { "cell_type": "markdown", - "id": "d0f24483", + "id": "19019bbb", "metadata": { "editable": true }, @@ -2498,7 +2501,7 @@ }, { "cell_type": "markdown", - "id": "be364340", + "id": "c7f1b6d0", "metadata": { "editable": true }, @@ -2509,7 +2512,7 @@ { "cell_type": "code", "execution_count": 24, - "id": "fc9dfe70", + "id": "62a22d27", "metadata": { "collapsed": false, "editable": true @@ -2524,7 +2527,7 @@ { "cell_type": "code", "execution_count": 25, - "id": "27fe4303", + "id": "3943a3d6", "metadata": { "collapsed": false, "editable": true diff --git a/doc/src/week44/Programs/DataFiles/grade.dot b/doc/src/week44/Programs/DataFiles/grade.dot new file mode 100644 index 000000000..dfcc5ea10 --- /dev/null +++ b/doc/src/week44/Programs/DataFiles/grade.dot @@ -0,0 +1,13 @@ +digraph Tree { +node [shape=box, style="filled, rounded", color="black", fontname="helvetica"] ; +edge [fontname="helvetica"] ; +0 [label="X[0] <= 0.5\ngini = 0.48\nsamples = 10\nvalue = [6, 4]", fillcolor="#f6d5bd"] ; +1 [label="X[5] <= 0.5\ngini = 0.444\nsamples = 6\nvalue = [2, 4]", fillcolor="#9ccef2"] ; +0 -> 1 [labeldistance=2.5, labelangle=45, headlabel="True"] ; +2 [label="gini = 0.0\nsamples = 2\nvalue = [2, 0]", fillcolor="#e58139"] ; +1 -> 2 ; +3 [label="gini = 0.0\nsamples = 4\nvalue = [0, 4]", fillcolor="#399de5"] ; +1 -> 3 ; +4 [label="gini = 0.0\nsamples = 4\nvalue = [4, 0]", fillcolor="#e58139"] ; +0 -> 4 [labeldistance=2.5, labelangle=-45, headlabel="False"] ; +} \ No newline at end of file diff --git a/doc/src/week44/Programs/DataFiles/grades.csv b/doc/src/week44/Programs/DataFiles/grades.csv new file mode 100644 index 000000000..2f3e97e30 --- /dev/null +++ b/doc/src/week44/Programs/DataFiles/grades.csv @@ -0,0 +1,13 @@ +Trend,Sleep,Studied,Grade +1,0,1,1 +0,1,0,0 +1,0,1,1 +1,1,1,1 +0,0,1,0 +1,0,0,0 +0,1,1,0 +0,0,1,0 +1,0,0,0 +1,1,1,1 + + diff --git a/doc/src/week44/Programs/DataFiles/grades.png b/doc/src/week44/Programs/DataFiles/grades.png new file mode 100644 index 000000000..aa8851b18 Binary files /dev/null and b/doc/src/week44/Programs/DataFiles/grades.png differ diff --git a/doc/src/week44/Programs/grades.csv b/doc/src/week44/Programs/grades.csv new file mode 100644 index 000000000..634017e6e --- /dev/null +++ b/doc/src/week44/Programs/grades.csv @@ -0,0 +1,13 @@ +Trend,Sleep,Studied,Grade, +1,0,1,1, +0,1,0,0, +1,0,1,1, +1,1,1,1, +0,0,1,0, +1,0,0,0, +0,1,1,0, +0,0,1,0, +1,0,0,0, +1,1,1,1, + + diff --git a/doc/src/week44/Programs/grades.py b/doc/src/week44/Programs/grades.py new file mode 100644 index 000000000..fc8f00c38 --- /dev/null +++ b/doc/src/week44/Programs/grades.py @@ -0,0 +1,70 @@ +# Common imports +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.tree import DecisionTreeClassifier +from sklearn.model_selection import train_test_split +from sklearn.tree import export_graphviz +from sklearn.preprocessing import StandardScaler, OneHotEncoder +from sklearn.compose import ColumnTransformer +from IPython.display import Image +from pydot import graph_from_dot_data +import os + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("grades.csv"),'r') + +# Read the experimental data with Pandas +from IPython.display import display +grades = pd.read_csv(infile)#, names = ('Trend','Sleep','Studied','Grade')) +grades = pd.DataFrame(grades) +display(grades) +# Features and targets +X = grades.loc[:, grades.columns != 'Grade'].values +y = grades.loc[:, grades.columns == 'Grade'].values +print(X) +# Create the encoder. +encoder = OneHotEncoder(handle_unknown="ignore") +# Assume for simplicity all features are categorical. +encoder.fit(X) +# Apply the encoder. +X = encoder.transform(X) +# Then do a Classification tree +tree_clf = DecisionTreeClassifier(max_depth=2) +tree_clf.fit(X, y) +print("Train set accuracy with Decision Tree: {:.2f}".format(tree_clf.score(X,y))) +#transfer to a decision tree graph +export_graphviz( + tree_clf, + out_file="DataFiles/grade.dot", + rounded=True, + filled=True +) +cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png' +os.system(cmd) + + +#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] diff --git a/doc/src/week44/week44.do.txt b/doc/src/week44/week44.do.txt index f6dc433a1..9b7c7063e 100644 --- a/doc/src/week44/week44.do.txt +++ b/doc/src/week44/week44.do.txt @@ -734,20 +734,19 @@ infile = open(data_path("grades.csv"),'r') # Read the experimental data with Pandas from IPython.display import display -grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade')) +grades = pd.read_csv(infile) grades = pd.DataFrame(grades) - +display(grades) # Features and targets X = grades.loc[:, grades.columns != 'Grade'].values y = grades.loc[:, grades.columns == 'Grade'].values - +print(X) # Create the encoder. encoder = OneHotEncoder(handle_unknown="ignore") # Assume for simplicity all features are categorical. encoder.fit(X) # Apply the encoder. X = encoder.transform(X) -print(X) # Then do a Classification tree tree_clf = DecisionTreeClassifier(max_depth=2) tree_clf.fit(X, y) @@ -761,6 +760,11 @@ export_graphviz( ) cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png' os.system(cmd) + + +#data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +#df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] + !ec