diff --git a/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html b/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html index 8136086bf..4591c0dc0 100644 --- a/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html +++ b/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html @@ -67,30 +67,32 @@ Automatically generated HTML file from DocOnce source ('A Classification Tree', 2, None, '___sec12'), ('Growing a classification tree', 2, None, '___sec13'), ('Classification tree, how to split nodes', 2, None, '___sec14'), - ('The CART (Classification and Regression Tree) algorithm', + ('Visualizing the Tree, Classification', 2, None, '___sec15'), + ('Visualizing the Tree, The Moons', 2, None, '___sec16'), + ('Computing the Gini index', 2, None, '___sec17'), + ('Simple Python Code to read in Data', 2, None, '___sec18'), + ('Computing the Gini Factor', 2, None, '___sec19'), + ('Entropy and the ID3 algorithm', 2, None, '___sec20'), + ('Implementing the ID3 Algorithm', 2, None, '___sec21'), + ('Cancer Data again now with Decision Trees and other Methods', 2, None, - '___sec15'), - ('Entropy and the ID3 algorithm', 2, None, '___sec16'), - ('Implementing the ID3 Algorithm', 2, None, '___sec17'), - ('Cancer Data again now with Decision Trees', - 2, - None, - '___sec18'), - ('Another example, the moons again', 2, None, '___sec19'), - ('Playing around with regions', 2, None, '___sec20'), - ('Regression trees', 2, None, '___sec21'), - ('Final regressor code', 2, None, '___sec22'), - ('Pros and cons of trees, pros', 2, None, '___sec23'), - ('Disadvantages', 2, None, '___sec24'), - ('Bagging', 2, None, '___sec25'), - ('Simple example, head or tail', 2, None, '___sec26'), - ('Random forests', 2, None, '___sec27'), - ('A simple scikit-learn example', 2, None, '___sec28'), - ('Please, not the moons again!', 2, None, '___sec29'), - ('Bagging examples', 2, None, '___sec30'), - ('Then random forests', 2, None, '___sec31'), - ('Boosting and more', 2, None, '___sec32')]} + '___sec22'), + ('Another example, the moons again', 2, None, '___sec23'), + ('Playing around with regions', 2, None, '___sec24'), + ('Regression trees', 2, None, '___sec25'), + ('Final regressor code', 2, None, '___sec26'), + ('Pros and cons of trees, pros', 2, None, '___sec27'), + ('Disadvantages', 2, None, '___sec28'), + ('Bagging', 2, None, '___sec29'), + ('More bagging', 2, None, '___sec30'), + ('Simple example, head or tail', 2, None, '___sec31'), + ('Bagging Example', 2, None, '___sec32'), + ('Random forests', 2, None, '___sec33'), + ('A simple scikit-learn example', 2, None, '___sec34'), + ('Please, not the moons again!', 2, None, '___sec35'), + ('Bagging examples', 2, None, '___sec36'), + ('Then random forests', 2, None, '___sec37')]} end of tocinfo -->
@@ -143,24 +145,29 @@ MathJax.Hub.Config({-
@@ -219,7 +226,7 @@ MathJax.Hub.Config({
-The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3. - +
-
from random import seed
-from random import randrange
-from csv import reader
-
-# Load a CSV file
-def load_csv(filename):
- file = open(filename, "rb")
- lines = reader(file)
- dataset = list(lines)
- return dataset
-
-# Convert string column to float
-def str_column_to_float(dataset, column):
- for row in dataset:
- row[column] = float(row[column].strip())
-
-# Split a dataset into k folds
-def cross_validation_split(dataset, n_folds):
- dataset_split = list()
- dataset_copy = list(dataset)
- fold_size = int(len(dataset) / n_folds)
- for i in range(n_folds):
- fold = list()
- while len(fold) < fold_size:
- index = randrange(len(dataset_copy))
- fold.append(dataset_copy.pop(index))
- dataset_split.append(fold)
- return dataset_split
-
-# Calculate accuracy percentage
-def accuracy_metric(actual, predicted):
- correct = 0
- for i in range(len(actual)):
- if actual[i] == predicted[i]:
- correct += 1
- return correct / float(len(actual)) * 100.0
-
-# Evaluate an algorithm using a cross validation split
-def evaluate_algorithm(dataset, algorithm, n_folds, *args):
- folds = cross_validation_split(dataset, n_folds)
- scores = list()
- for fold in folds:
- train_set = list(folds)
- train_set.remove(fold)
- train_set = sum(train_set, [])
- test_set = list()
- for row in fold:
- row_copy = list(row)
- test_set.append(row_copy)
- row_copy[-1] = None
- predicted = algorithm(train_set, test_set, *args)
- actual = [row[-1] for row in fold]
- accuracy = accuracy_metric(actual, predicted)
- scores.append(accuracy)
- return scores
-
-# Split a dataset based on an attribute and an attribute value
-def test_split(index, value, dataset):
- left, right = list(), list()
- for row in dataset:
- if row[index] < value:
- left.append(row)
- else:
- right.append(row)
- return left, right
-
-# Calculate the Gini index for a split dataset
-def gini_index(groups, classes):
- # count all samples at split point
- n_instances = float(sum([len(group) for group in groups]))
- # sum weighted Gini index for each group
- gini = 0.0
- for group in groups:
- size = float(len(group))
- # avoid divide by zero
- if size == 0:
- continue
- score = 0.0
- # score the group based on the score for each class
- for class_val in classes:
- p = [row[-1] for row in group].count(class_val) / size
- score += p * p
- # weight the group score by its relative size
- gini += (1.0 - score) * (size / n_instances)
- return gini
-
-# Select the best split point for a dataset
-def get_split(dataset):
- class_values = list(set(row[-1] for row in dataset))
- b_index, b_value, b_score, b_groups = 999, 999, 999, None
- for index in range(len(dataset[0])-1):
- for row in dataset:
- groups = test_split(index, row[index], dataset)
- gini = gini_index(groups, class_values)
- if gini < b_score:
- b_index, b_value, b_score, b_groups = index, row[index], gini, groups
- return {'index':b_index, 'value':b_value, 'groups':b_groups}
-
-# Create a terminal node value
-def to_terminal(group):
- outcomes = [row[-1] for row in group]
- return max(set(outcomes), key=outcomes.count)
-
-# Create child splits for a node or make terminal
-def split(node, max_depth, min_size, depth):
- left, right = node['groups']
- del(node['groups'])
- # check for a no split
- if not left or not right:
- node['left'] = node['right'] = to_terminal(left + right)
- return
- # check for max depth
- if depth >= max_depth:
- node['left'], node['right'] = to_terminal(left), to_terminal(right)
- return
- # process left child
- if len(left) <= min_size:
- node['left'] = to_terminal(left)
- else:
- node['left'] = get_split(left)
- split(node['left'], max_depth, min_size, depth+1)
- # process right child
- if len(right) <= min_size:
- node['right'] = to_terminal(right)
- else:
- node['right'] = get_split(right)
- split(node['right'], max_depth, min_size, depth+1)
-
-# Build a decision tree
-def build_tree(train, max_depth, min_size):
- root = get_split(train)
- split(root, max_depth, min_size, 1)
- return root
-
-# Make a prediction with a decision tree
-def predict(node, row):
- if row[node['index']] < node['value']:
- if isinstance(node['left'], dict):
- return predict(node['left'], row)
- else:
- return node['left']
- else:
- if isinstance(node['right'], dict):
- return predict(node['right'], row)
- else:
- return node['right']
-
-# Classification and Regression Tree Algorithm
-def decision_tree(train, test, max_depth, min_size):
- tree = build_tree(train, max_depth, min_size)
- predictions = list()
- for row in test:
- prediction = predict(tree, row)
- predictions.append(prediction)
- return(predictions)
-
-# Test CART
-seed(1)
-# load and prepare data
-filename = 'DataFiles/rideclass.csv'
-dataset = load_csv(filename)
-# convert string attributes to integers
-for i in range(len(dataset[0])):
- str_column_to_float(dataset, i)
-# evaluate algorithm
-n_folds = 5
-max_depth = 5
-min_size = 10
-scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
-print('Scores: %s' % scores)
-print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+import os
+from sklearn.datasets import load_breast_cancer
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.model_selection import train_test_split
+from sklearn.metrics import confusion_matrix
+from sklearn.tree import export_graphviz
+
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import pandas as pd
+import numpy as np
+
+
+cancer = load_breast_cancer()
+X = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+print(X)
+y = pd.Categorical.from_codes(cancer.target, cancer.target_names)
+y = pd.get_dummies(y)
+print(y)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/cancer.dot",
+ feature_names=cancer.feature_names,
+ class_names=cancer.target_names,
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png'
+os.system(cmd)
@@ -382,7 +247,7 @@ scores = evaluate_algorithm(dataset, decisio
-ID3, learns decision trees by constructing -them topdown, beginning with the question which attribute should be tested at the root of the tree? -
# Common imports
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.datasets import make_moons
+from sklearn.tree import export_graphviz
+from pydot import graph_from_dot_data
+import pandas as pd
+import os
-The ID3 algorithm selects, which attribute to test at each node in the
-tree.
-
-
-We would like to select the attribute that is most useful for classifying
-examples.
-
-
-What is a good quantitative measure of the worth of an attribute?
-
-
-Information gain measures how well a given attribute separates the
-training examples according to their target classification.
-
-
-The ID3 algorithm uses this information gain measure to select among the candidate
-attributes at each step while growing the tree.
+np.random.seed(42)
+X, y = make_moons(n_samples=100, noise=0.25, random_state=53)
+X_train, X_test, y_train, y_test = train_test_split(X,y,random_state=0)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/moons.dot",
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/moons.dot -o DataFiles/moons.png'
+os.system(cmd)
+
@@ -235,7 +238,7 @@ attributes at each step while growing the tree.
-more text to come here, material presented during lecture Friday Oct 25. +The example we will look at is a classical one in many Machine +Learning applications. Based on various meteorological features, we +have several so-called attributes which decide whether we at the end +will do some outdoor activity like skiing, going for a bike ride etc +etc. The table here contains the feautures outlook, temperature, +humidity and wind. The target or output is whether we ride +(True=1) or whether we do something else that day (False=0). The +attributes for each feature are then sunny, overcast and rain for the +outlook, hot, cold and mild for temperature, high and normal for +humidity and weak and strong for wind. +
+The table here summarizes the various attributes and + +
| Day | Outlook | Temperature | Humidity | Wind | Ride |
| 1 | Sunny | Hot | High | Weak | 0 |
| 2 | Sunny | Hot | High | Strong | 1 |
| 3 | Overcast | Hot | High | Weak | 1 |
| 4 | Rain | Mild | High | Weak | 1 |
| 5 | Rain | Cool | Normal | Weak | 1 |
| 6 | Rain | Cool | Normal | Strong | 0 |
| 7 | Overcast | Cool | Normal | Strong | 1 |
| 8 | Sunny | Mild | High | Weak | 0 |
| 9 | Sunny | Cool | Normal | Weak | 1 |
| 10 | Rain | Mild | Normal | Weak | 1 |
| 11 | Sunny | Mild | Normal | Strong | 1 |
| 12 | Overcast | Mild | High | Strong | 1 |
| 13 | Overcast | Hot | Normal | Weak | 1 |
| 14 | Rain | Mild | High | Strong | 0 |
@@ -207,7 +251,7 @@ MathJax.Hub.Config({
-
import matplotlib.pyplot as plt
+# Common imports
import numpy as np
-from sklearn.model_selection import train_test_split
-from sklearn.datasets import load_breast_cancer
-from sklearn.svm import SVC
-from sklearn.linear_model import LogisticRegression
-from sklearn.tree import DecisionTreeClassifier
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.preprocessing import StandardScaler, OneHotEncoder
+from sklearn.compose import ColumnTransformer
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import os
-# Load the data
-cancer = load_breast_cancer()
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
-X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
-print(X_train.shape)
-print(X_test.shape)
-# Logistic Regression
-logreg = LogisticRegression(solver='lbfgs')
-logreg.fit(X_train, y_train)
-print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
-# Support vector machine
-svm = SVC(gamma='auto', C=100)
-svm.fit(X_train, y_train)
-print("Test set accuracy with SVM: {:.2f}".format(svm.score(X_test,y_test)))
-# Decision Trees
-deep_tree_clf = DecisionTreeClassifier(max_depth=None)
-deep_tree_clf.fit(X_train, y_train)
-print("Test set accuracy with Decision Trees: {:.2f}".format(deep_tree_clf.score(X_test,y_test)))
-#now scale the data
-from sklearn.preprocessing import StandardScaler
-scaler = StandardScaler()
-scaler.fit(X_train)
-X_train_scaled = scaler.transform(X_train)
-X_test_scaled = scaler.transform(X_test)
-# Logistic Regression
-logreg.fit(X_train_scaled, y_train)
-print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-# Support Vector Machine
-svm.fit(X_train_scaled, y_train)
-print("Test set accuracy SVM with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-# Decision Trees
-deep_tree_clf.fit(X_train_scaled, y_train)
-print("Test set accuracy with Decision Trees and scaled data: {:.2f}".format(deep_tree_clf.score(X_test_scaled,y_test)))
+if not os.path.exists(PROJECT_ROOT_DIR):
+ os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+ os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+ os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+ return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+ return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+ plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("ride.csv"),'r')
+
+# Read the experimental data with Pandas
+from IPython.display import display
+ridedata = pd.read_csv(infile,names = ('Outlook','Temperature','Humidity','Wind','Ride'))
+ridedata = pd.DataFrame(ridedata)
+display(ridedata)
+# Features and targets
+X = ridedata.loc[:, ridedata.columns != 'Ride'].values
+display(X)
+y = ridedata.loc[:, ridedata.columns == 'Ride'].values
+display(y)
+# Categorical variables to one-hot's
+onehotencoder = OneHotEncoder(categories="auto")
+
+X = ColumnTransformer([("", onehotencoder)]).fit_transform(X)
+y.shape
+
+display(X)
+display(y)
@@ -248,7 +268,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
+The above functions (gini, entropy and misclassification error) are +important components of the so-called CART algorithm. We will discuss +this algorithm below after we have discussed the information gain +algorithm ID3. + +
+In the example here we have converted all our attributes into numerical values \( 0,1,2 \) etc. +
-
from __future__ import division, print_function, unicode_literals
+# Split a dataset based on an attribute and an attribute value
+def test_split(index, value, dataset):
+ left, right = list(), list()
+ for row in dataset:
+ if row[index] < value:
+ left.append(row)
+ else:
+ right.append(row)
+ return left, right
+
+# Calculate the Gini index for a split dataset
+def gini_index(groups, classes):
+ # count all samples at split point
+ n_instances = float(sum([len(group) for group in groups]))
+ # sum weighted Gini index for each group
+ gini = 0.0
+ for group in groups:
+ size = float(len(group))
+ # avoid divide by zero
+ if size == 0:
+ continue
+ score = 0.0
+ # score the group based on the score for each class
+ for class_val in classes:
+ p = [row[-1] for row in group].count(class_val) / size
+ score += p * p
+ # weight the group score by its relative size
+ gini += (1.0 - score) * (size / n_instances)
+ return gini
-# Common imports
-import numpy as np
-import os
+# Select the best split point for a dataset
+def get_split(dataset):
+ class_values = list(set(row[-1] for row in dataset))
+ b_index, b_value, b_score, b_groups = 999, 999, 999, None
+ for index in range(len(dataset[0])-1):
+ for row in dataset:
+ groups = test_split(index, row[index], dataset)
+ gini = gini_index(groups, class_values)
+ print('X%d < %.3f Gini=%.3f' % ((index+1), row[index], gini))
+ if gini < b_score:
+ b_index, b_value, b_score, b_groups = index, row[index], gini, groups
+ return {'index':b_index, 'value':b_value, 'groups':b_groups}
+
+dataset = [[0,0,0,0,0],
+ [0,0,0,1,1],
+ [1,0,0,0,1],
+ [2,1,0,0,1],
+ [2,2,1,0,1],
+ [2,2,1,1,0],
+ [1,2,1,1,1],
+ [0,1,0,0,0],
+ [0,2,1,0,1],
+ [2,1,1,0,1],
+ [0,1,1,1,1],
+ [1,1,0,1,1],
+ [1,0,1,0,1],
+ [2,1,0,1,0]]
-# to make this notebook's output stable across runs
-np.random.seed(42)
-
-# To plot pretty figures
-import matplotlib
-import matplotlib.pyplot as plt
-from matplotlib.colors import ListedColormap
-plt.rcParams['axes.labelsize'] = 14
-plt.rcParams['xtick.labelsize'] = 12
-plt.rcParams['ytick.labelsize'] = 12
-
-
-from sklearn.svm import SVC
-from sklearn import datasets
-from sklearn.tree import DecisionTreeClassifier
-from sklearn.datasets import make_moons
-from sklearn.tree import export_graphviz
-
-Xm, ym = make_moons(n_samples=100, noise=0.25, random_state=53)
-
-deep_tree_clf1 = DecisionTreeClassifier(random_state=42)
-deep_tree_clf2 = DecisionTreeClassifier(min_samples_leaf=4, random_state=42)
-deep_tree_clf1.fit(Xm, ym)
-deep_tree_clf2.fit(Xm, ym)
-
-
-def plot_decision_boundary(clf, X, y, axes=[0, 7.5, 0, 3], iris=True, legend=False, plot_training=True):
- x1s = np.linspace(axes[0], axes[1], 100)
- x2s = np.linspace(axes[2], axes[3], 100)
- x1, x2 = np.meshgrid(x1s, x2s)
- X_new = np.c_[x1.ravel(), x2.ravel()]
- y_pred = clf.predict(X_new).reshape(x1.shape)
- custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
- plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
- if not iris:
- custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
- plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
- if plot_training:
- plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", label="Iris-Setosa")
- plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", label="Iris-Versicolor")
- plt.plot(X[:, 0][y==2], X[:, 1][y==2], "g^", label="Iris-Virginica")
- plt.axis(axes)
- if iris:
- plt.xlabel("Petal length", fontsize=14)
- plt.ylabel("Petal width", fontsize=14)
- else:
- plt.xlabel(r"$x_1$", fontsize=18)
- plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
- if legend:
- plt.legend(loc="lower right", fontsize=14)
-plt.figure(figsize=(11, 4))
-plt.subplot(121)
-plot_decision_boundary(deep_tree_clf1, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
-plt.title("No restrictions", fontsize=16)
-plt.subplot(122)
-plot_decision_boundary(deep_tree_clf2, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
-plt.title("min_samples_leaf = {}".format(deep_tree_clf2.min_samples_leaf), fontsize=14)
-plt.show()
+split = get_split(dataset)
+print('Split: [X%d < %.3f]' % ((split['index']+1), split['value']))
@@ -271,7 +284,7 @@ plt.show()
+ID3, learns decision trees by constructing +them topdown, beginning with the question which attribute should be tested at the root of the tree? - -
np.random.seed(6)
-Xs = np.random.rand(100, 2) - 0.5
-ys = (Xs[:, 0] > 0).astype(np.float32) * 2
+
+- Each instance attribute is evaluated using a statistical test to determine how well it alone classifies the training examples.
+- The best attribute is selected and used as the test at the root node of the tree.
+- A descendant of the root node is then created for each possible value of this attribute.
+- Training examples are sorted to the appropriate descendant node.
+- The entire process is then repeated using the training examples associated with each descendant node to select the best attribute to test at that point in the tree.
+- This forms a greedy search for an acceptable decision tree, in which the algorithm never backtracks to reconsider earlier choices.
+
-angle = np.pi / 4
-rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
-Xsr = Xs.dot(rotation_matrix)
+The ID3 algorithm selects, which attribute to test at each node in the
+tree.
-tree_clf_s = DecisionTreeClassifier(random_state=42)
-tree_clf_s.fit(Xs, ys)
-tree_clf_sr = DecisionTreeClassifier(random_state=42)
-tree_clf_sr.fit(Xsr, ys)
+
+We would like to select the attribute that is most useful for classifying
+examples.
-plt.figure(figsize=(11, 4))
-plt.subplot(121)
-plot_decision_boundary(tree_clf_s, Xs, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
-plt.subplot(122)
-plot_decision_boundary(tree_clf_sr, Xsr, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
+
+What is a good quantitative measure of the worth of an attribute?
+
+
+Information gain measures how well a given attribute separates the
+training examples according to their target classification.
+
+
+The ID3 algorithm uses this information gain measure to select among the candidate
+attributes at each step while growing the tree.
-plt.show()
-
@@ -227,7 +242,7 @@ plt.show()
+import re +import math +from collections import deque - -
# Quadratic training set + noise
-np.random.seed(42)
-m = 200
-X = np.random.rand(m, 1)
-y = 4 * (X - 0.5) ** 2
-y = y + np.random.randn(m, 1) / 10
-+ + + + + - -
from sklearn.tree import DecisionTreeRegressor
+
+class Node(object):
+ def __init__(self):
+ self.value = None
+ self.next = None
+ self.childs = None
+
+
+
+
+class DecisionTree(object):
+ def __init__(self, sample, attributes, labels):
+ self.sample = sample
+ self.attributes = attributes
+ self.labels = labels
+ self.labelCodes = None
+ self.labelCodesCount = None
+ self.initLabelCodes()
+ # print(self.labelCodes)
+ self.root = None
+ self.entropy = self.getEntropy([x for x in range(len(self.labels))])
+
+
+ def initLabelCodes(self):
+ self.labelCodes = []
+ self.labelCodesCount = []
+ for l in self.labels:
+ if l not in self.labelCodes:
+ self.labelCodes.append(l)
+ self.labelCodesCount.append(0)
+ self.labelCodesCount[self.labelCodes.index(l)] += 1
+
+
+ def getLabelCodeId(self, sampleId):
+ return self.labelCodes.index(self.labels[sampleId])
+
+
+ def getAttributeValues(self, sampleIds, attributeId):
+ vals = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in vals:
+ vals.append(val)
+ # print(vals)
+ return vals
+
+
+ def getEntropy(self, sampleIds):
+ entropy = 0
+ labelCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCount[self.getLabelCodeId(sid)] += 1
+ # print("-ge", labelCount)
+ for lv in labelCount:
+ # print(lv)
+ if lv != 0:
+ entropy += -lv/len(sampleIds) * math.log(lv/len(sampleIds), 2)
+ else:
+ entropy += 0
+ return entropy
+
+
+ def getDominantLabel(self, sampleIds):
+ labelCodesCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCodesCount[self.labelCodes.index(self.labels[sid])] += 1
+ return self.labelCodes[labelCodesCount.index(max(labelCodesCount))]
+
+
+ def getInformationGain(self, sampleIds, attributeId):
+ gain = self.getEntropy(sampleIds)
+ attributeVals = []
+ attributeValsCount = []
+ attributeValsIds = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in attributeVals:
+ attributeVals.append(val)
+ attributeValsCount.append(0)
+ attributeValsIds.append([])
+ vid = attributeVals.index(val)
+ attributeValsCount[vid] += 1
+ attributeValsIds[vid].append(sid)
+ # print("-gig", self.attributes[attributeId])
+ for vc, vids in zip(attributeValsCount, attributeValsIds):
+ # print("-gig", vids)
+ gain -= vc/len(sampleIds) * self.getEntropy(vids)
+ return gain
+
+
+ def getAttributeMaxInformationGain(self, sampleIds, attributeIds):
+ attributesEntropy = [0] * len(attributeIds)
+ for i, attId in zip(range(len(attributeIds)), attributeIds):
+ attributesEntropy[i] = self.getInformationGain(sampleIds, attId)
+ maxId = attributeIds[attributesEntropy.index(max(attributesEntropy))]
+ return self.attributes[maxId], maxId
+
+
+ def isSingleLabeled(self, sampleIds):
+ label = self.labels[sampleIds[0]]
+ for sid in sampleIds:
+ if self.labels[sid] != label:
+ return False
+ return True
+
+
+ def getLabel(self, sampleId):
+ return self.labels[sampleId]
+
+
+ def id3(self):
+ sampleIds = [x for x in range(len(self.sample))]
+ attributeIds = [x for x in range(len(self.attributes))]
+ self.root = self.id3Recv(sampleIds, attributeIds, self.root)
+
+
+ def id3Recv(self, sampleIds, attributeIds, root):
+ root = Node() # Initialize current root
+ if self.isSingleLabeled(sampleIds):
+ root.value = self.labels[sampleIds[0]]
+ return root
+ # print(attributeIds)
+ if len(attributeIds) == 0:
+ root.value = self.getDominantLabel(sampleIds)
+ return root
+ bestAttrName, bestAttrId = self.getAttributeMaxInformationGain(
+ sampleIds, attributeIds)
+ # print(bestAttrName)
+ root.value = bestAttrName
+ root.childs = [] # Create list of children
+ for value in self.getAttributeValues(sampleIds, bestAttrId):
+ # print(value)
+ child = Node()
+ child.value = value
+ root.childs.append(child) # Append new child node to current
+ # root
+ childSampleIds = []
+ for sid in sampleIds:
+ if self.sample[sid][bestAttrId] == value:
+ childSampleIds.append(sid)
+ if len(childSampleIds) == 0:
+ child.next = self.getDominantLabel(sampleIds)
+ else:
+ # print(bestAttrName, bestAttrId)
+ # print(attributeIds)
+ if len(attributeIds) > 0 and bestAttrId in attributeIds:
+ toRemove = attributeIds.index(bestAttrId)
+ attributeIds.pop(toRemove)
+ child.next = self.id3Recv(
+ childSampleIds, attributeIds, child.next)
+ return root
+
+
+ def printTree(self):
+ if self.root:
+ roots = deque()
+ roots.append(self.root)
+ while len(roots) > 0:
+ root = roots.popleft()
+ print(root.value)
+ if root.childs:
+ for child in root.childs:
+ print('({})'.format(child.value))
+ roots.append(child.next)
+ elif root.next:
+ print(root.next)
+
+
+def test():
+ f = open('DataFiles/rideclass.csv')
+ attributes = f.readline().split(',')
+ attributes = attributes[1:len(attributes)-1]
+ print(attributes)
+ sample = f.readlines()
+ f.close()
+ for i in range(len(sample)):
+ sample[i] = re.sub('\d+,', '', sample[i])
+ sample[i] = sample[i].strip().split(',')
+ labels = []
+ for s in sample:
+ labels.append(s.pop())
+ # print(sample)
+ # print(labels)
+ decisionTree = DecisionTree(sample, attributes, labels)
+ print("System entropy {}".format(decisionTree.entropy))
+ decisionTree.id3()
+ decisionTree.printTree()
+
+
+if __name__ == '__main__':
+ test()
-tree_reg = DecisionTreeRegressor(max_depth=2, random_state=42)
-tree_reg.fit(X, y)
-
@@ -221,7 +415,7 @@ tree_reg.fit(X, y)
-
from sklearn.tree import DecisionTreeRegressor
+import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.datasets import load_breast_cancer
+from sklearn.svm import SVC
+from sklearn.linear_model import LogisticRegression
+from sklearn.tree import DecisionTreeClassifier
-tree_reg1 = DecisionTreeRegressor(random_state=42, max_depth=2)
-tree_reg2 = DecisionTreeRegressor(random_state=42, max_depth=3)
-tree_reg1.fit(X, y)
-tree_reg2.fit(X, y)
+# Load the data
+cancer = load_breast_cancer()
-def plot_regression_predictions(tree_reg, X, y, axes=[0, 1, -0.2, 1], ylabel="$y$"):
- x1 = np.linspace(axes[0], axes[1], 500).reshape(-1, 1)
- y_pred = tree_reg.predict(x1)
- plt.axis(axes)
- plt.xlabel("$x_1$", fontsize=18)
- if ylabel:
- plt.ylabel(ylabel, fontsize=18, rotation=0)
- plt.plot(X, y, "b.")
- plt.plot(x1, y_pred, "r.-", linewidth=2, label=r"$\hat{y}$")
-
-plt.figure(figsize=(11, 4))
-plt.subplot(121)
-plot_regression_predictions(tree_reg1, X, y)
-for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
- plt.plot([split, split], [-0.2, 1], style, linewidth=2)
-plt.text(0.21, 0.65, "Depth=0", fontsize=15)
-plt.text(0.01, 0.2, "Depth=1", fontsize=13)
-plt.text(0.65, 0.8, "Depth=1", fontsize=13)
-plt.legend(loc="upper center", fontsize=18)
-plt.title("max_depth=2", fontsize=14)
-
-plt.subplot(122)
-plot_regression_predictions(tree_reg2, X, y, ylabel=None)
-for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
- plt.plot([split, split], [-0.2, 1], style, linewidth=2)
-for split in (0.0458, 0.1298, 0.2873, 0.9040):
- plt.plot([split, split], [-0.2, 1], "k:", linewidth=1)
-plt.text(0.3, 0.5, "Depth=2", fontsize=13)
-plt.title("max_depth=3", fontsize=14)
-
-plt.show()
-
-
-
-
-
tree_reg1 = DecisionTreeRegressor(random_state=42)
-tree_reg2 = DecisionTreeRegressor(random_state=42, min_samples_leaf=10)
-tree_reg1.fit(X, y)
-tree_reg2.fit(X, y)
-
-x1 = np.linspace(0, 1, 500).reshape(-1, 1)
-y_pred1 = tree_reg1.predict(x1)
-y_pred2 = tree_reg2.predict(x1)
-
-plt.figure(figsize=(11, 4))
-
-plt.subplot(121)
-plt.plot(X, y, "b.")
-plt.plot(x1, y_pred1, "r.-", linewidth=2, label=r"$\hat{y}$")
-plt.axis([0, 1, -0.2, 1.1])
-plt.xlabel("$x_1$", fontsize=18)
-plt.ylabel("$y$", fontsize=18, rotation=0)
-plt.legend(loc="upper center", fontsize=18)
-plt.title("No restrictions", fontsize=14)
-
-plt.subplot(122)
-plt.plot(X, y, "b.")
-plt.plot(x1, y_pred2, "r.-", linewidth=2, label=r"$\hat{y}$")
-plt.axis([0, 1, -0.2, 1.1])
-plt.xlabel("$x_1$", fontsize=18)
-plt.title("min_samples_leaf={}".format(tree_reg2.min_samples_leaf), fontsize=14)
-
-plt.show()
+X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
+print(X_train.shape)
+print(X_test.shape)
+# Logistic Regression
+logreg = LogisticRegression(solver='lbfgs')
+logreg.fit(X_train, y_train)
+print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
+# Support vector machine
+svm = SVC(gamma='auto', C=100)
+svm.fit(X_train, y_train)
+print("Test set accuracy with SVM: {:.2f}".format(svm.score(X_test,y_test)))
+# Decision Trees
+deep_tree_clf = DecisionTreeClassifier(max_depth=None)
+deep_tree_clf.fit(X_train, y_train)
+print("Test set accuracy with Decision Trees: {:.2f}".format(deep_tree_clf.score(X_test,y_test)))
+#now scale the data
+from sklearn.preprocessing import StandardScaler
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+# Logistic Regression
+logreg.fit(X_train_scaled, y_train)
+print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+# Support Vector Machine
+svm.fit(X_train_scaled, y_train)
+print("Test set accuracy SVM with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+# Decision Trees
+deep_tree_clf.fit(X_train_scaled, y_train)
+print("Test set accuracy with Decision Trees and scaled data: {:.2f}".format(deep_tree_clf.score(X_test_scaled,y_test)))
@@ -277,7 +255,7 @@ plt.show()
-
from __future__ import division, print_function, unicode_literals
+# Common imports
+import numpy as np
+import os
+
+# to make this notebook's output stable across runs
+np.random.seed(42)
+
+# To plot pretty figures
+import matplotlib
+import matplotlib.pyplot as plt
+from matplotlib.colors import ListedColormap
+plt.rcParams['axes.labelsize'] = 14
+plt.rcParams['xtick.labelsize'] = 12
+plt.rcParams['ytick.labelsize'] = 12
+
+
+from sklearn.svm import SVC
+from sklearn import datasets
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.datasets import make_moons
+from sklearn.tree import export_graphviz
+
+Xm, ym = make_moons(n_samples=100, noise=0.25, random_state=53)
+
+deep_tree_clf1 = DecisionTreeClassifier(random_state=42)
+deep_tree_clf2 = DecisionTreeClassifier(min_samples_leaf=4, random_state=42)
+deep_tree_clf1.fit(Xm, ym)
+deep_tree_clf2.fit(Xm, ym)
+
+
+def plot_decision_boundary(clf, X, y, axes=[0, 7.5, 0, 3], iris=True, legend=False, plot_training=True):
+ x1s = np.linspace(axes[0], axes[1], 100)
+ x2s = np.linspace(axes[2], axes[3], 100)
+ x1, x2 = np.meshgrid(x1s, x2s)
+ X_new = np.c_[x1.ravel(), x2.ravel()]
+ y_pred = clf.predict(X_new).reshape(x1.shape)
+ custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
+ plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
+ if not iris:
+ custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
+ plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
+ if plot_training:
+ plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", label="Iris-Setosa")
+ plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", label="Iris-Versicolor")
+ plt.plot(X[:, 0][y==2], X[:, 1][y==2], "g^", label="Iris-Virginica")
+ plt.axis(axes)
+ if iris:
+ plt.xlabel("Petal length", fontsize=14)
+ plt.ylabel("Petal width", fontsize=14)
+ else:
+ plt.xlabel(r"$x_1$", fontsize=18)
+ plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
+ if legend:
+ plt.legend(loc="lower right", fontsize=14)
+plt.figure(figsize=(11, 4))
+plt.subplot(121)
+plot_decision_boundary(deep_tree_clf1, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
+plt.title("No restrictions", fontsize=16)
+plt.subplot(122)
+plot_decision_boundary(deep_tree_clf2, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
+plt.title("min_samples_leaf = {}".format(deep_tree_clf2.min_samples_leaf), fontsize=14)
+plt.show()
+
diff --git a/doc/pub/DecisionTrees/html/._DecisionTrees-bs025.html b/doc/pub/DecisionTrees/html/._DecisionTrees-bs025.html index 199639975..9b12a47de 100644 --- a/doc/pub/DecisionTrees/html/._DecisionTrees-bs025.html +++ b/doc/pub/DecisionTrees/html/._DecisionTrees-bs025.html @@ -67,30 +67,32 @@ Automatically generated HTML file from DocOnce source ('A Classification Tree', 2, None, '___sec12'), ('Growing a classification tree', 2, None, '___sec13'), ('Classification tree, how to split nodes', 2, None, '___sec14'), - ('The CART (Classification and Regression Tree) algorithm', + ('Visualizing the Tree, Classification', 2, None, '___sec15'), + ('Visualizing the Tree, The Moons', 2, None, '___sec16'), + ('Computing the Gini index', 2, None, '___sec17'), + ('Simple Python Code to read in Data', 2, None, '___sec18'), + ('Computing the Gini Factor', 2, None, '___sec19'), + ('Entropy and the ID3 algorithm', 2, None, '___sec20'), + ('Implementing the ID3 Algorithm', 2, None, '___sec21'), + ('Cancer Data again now with Decision Trees and other Methods', 2, None, - '___sec15'), - ('Entropy and the ID3 algorithm', 2, None, '___sec16'), - ('Implementing the ID3 Algorithm', 2, None, '___sec17'), - ('Cancer Data again now with Decision Trees', - 2, - None, - '___sec18'), - ('Another example, the moons again', 2, None, '___sec19'), - ('Playing around with regions', 2, None, '___sec20'), - ('Regression trees', 2, None, '___sec21'), - ('Final regressor code', 2, None, '___sec22'), - ('Pros and cons of trees, pros', 2, None, '___sec23'), - ('Disadvantages', 2, None, '___sec24'), - ('Bagging', 2, None, '___sec25'), - ('Simple example, head or tail', 2, None, '___sec26'), - ('Random forests', 2, None, '___sec27'), - ('A simple scikit-learn example', 2, None, '___sec28'), - ('Please, not the moons again!', 2, None, '___sec29'), - ('Bagging examples', 2, None, '___sec30'), - ('Then random forests', 2, None, '___sec31'), - ('Boosting and more', 2, None, '___sec32')]} + '___sec22'), + ('Another example, the moons again', 2, None, '___sec23'), + ('Playing around with regions', 2, None, '___sec24'), + ('Regression trees', 2, None, '___sec25'), + ('Final regressor code', 2, None, '___sec26'), + ('Pros and cons of trees, pros', 2, None, '___sec27'), + ('Disadvantages', 2, None, '___sec28'), + ('Bagging', 2, None, '___sec29'), + ('More bagging', 2, None, '___sec30'), + ('Simple example, head or tail', 2, None, '___sec31'), + ('Bagging Example', 2, None, '___sec32'), + ('Random forests', 2, None, '___sec33'), + ('A simple scikit-learn example', 2, None, '___sec34'), + ('Please, not the moons again!', 2, None, '___sec35'), + ('Bagging examples', 2, None, '___sec36'), + ('Then random forests', 2, None, '___sec37')]} end of tocinfo --> @@ -143,24 +145,29 @@ MathJax.Hub.Config({
-
np.random.seed(6)
+Xs = np.random.rand(100, 2) - 0.5
+ys = (Xs[:, 0] > 0).astype(np.float32) * 2
-However, by aggregating many decision trees, using methods like bagging, random forests, and boosting, the predictive performance of trees can be substantially improved.
+angle = np.pi/4
+rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
+Xsr = Xs.dot(rotation_matrix)
+tree_clf_s = DecisionTreeClassifier(random_state=42)
+tree_clf_s.fit(Xs, ys)
+tree_clf_sr = DecisionTreeClassifier(random_state=42)
+tree_clf_sr.fit(Xsr, ys)
+
+plt.figure(figsize=(11, 4))
+plt.subplot(121)
+plot_decision_boundary(tree_clf_s, Xs, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
+plt.subplot(122)
+plot_decision_boundary(tree_clf_sr, Xsr, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
+
+plt.show()
+
@@ -214,6 +232,9 @@ However, by aggregating many decision trees, using methods like bagging, random
-The plain decision trees suffer from high -variance. This means that if we split the training data into two parts -at random, and fit a decision tree to both halves, the results that we -get could be quite different. In contrast, a procedure with low -variance will yield similar results if applied repeatedly to distinct -data sets; linear regression tends to have low variance, if the ratio -of \( n \) to \( p \) is moderately large. + +
# Quadratic training set + noise
+np.random.seed(42)
+m = 200
+X = np.random.rand(m, 1)
+y = 4 * (X - 0.5) ** 2
+y = y + np.random.randn(m, 1) / 10
+-Bootstrap aggregation, or just bagging, is a -general-purpose procedure for reducing the variance of a statistical -learning method. -
-Bagging typically results in improved accuracy -over prediction using a single tree. Unfortunately, however, it can be -difficult to interpret the resulting model. Recall that one of the -advantages of decision trees is the attractive and easily interpreted -diagram that results. - -
-However, when we bag a large number of trees, it is no longer -possible to represent the resulting statistical learning procedure -using a single tree, and it is no longer clear which variables are -most important to the procedure. Thus, bagging improves prediction -accuracy at the expense of interpretability. Although the collection -of bagged trees is much more difficult to interpret than a single -tree, one can obtain an overall summary of the importance of each -predictor using the MSE (for bagging regression trees) or the Gini -index (for bagging classification trees). In the case of bagging -regression trees, we can record the total amount that the MSE is -decreased due to splits over a given predictor, averaged over all \( B \) possible -trees. A large value indicates an important predictor. Similarly, in -the context of bagging classification trees, we can add up the total -amount that the Gini index is decreased by splits over a given -predictor, averaged over all \( B \) trees. + +
from sklearn.tree import DecisionTreeRegressor
+tree_reg = DecisionTreeRegressor(max_depth=2, random_state=42)
+tree_reg.fit(X, y)
+
@@ -239,6 +225,10 @@ predictor, averaged over all \( B \) trees.
-
heads_proba = 0.51
-coin_tosses = (np.random.rand(10000, 10) < heads_proba).astype(np.int32)
-cumulative_heads_ratio = np.cumsum(coin_tosses, axis=0) / np.arange(1, 10001).reshape(-1, 1)
-plt.figure(figsize=(8,3.5))
-plt.plot(cumulative_heads_ratio)
-plt.plot([0, 10000], [0.51, 0.51], "k--", linewidth=2, label="51%")
-plt.plot([0, 10000], [0.5, 0.5], "k-", label="50%")
-plt.xlabel("Number of coin tosses")
-plt.ylabel("Heads ratio")
-plt.legend(loc="lower right")
-plt.axis([0, 10000, 0.42, 0.58])
+from sklearn.tree import DecisionTreeRegressor
+
+tree_reg1 = DecisionTreeRegressor(random_state=42, max_depth=2)
+tree_reg2 = DecisionTreeRegressor(random_state=42, max_depth=3)
+tree_reg1.fit(X, y)
+tree_reg2.fit(X, y)
+
+def plot_regression_predictions(tree_reg, X, y, axes=[0, 1, -0.2, 1], ylabel="$y$"):
+ x1 = np.linspace(axes[0], axes[1], 500).reshape(-1, 1)
+ y_pred = tree_reg.predict(x1)
+ plt.axis(axes)
+ plt.xlabel("$x_1$", fontsize=18)
+ if ylabel:
+ plt.ylabel(ylabel, fontsize=18, rotation=0)
+ plt.plot(X, y, "b.")
+ plt.plot(x1, y_pred, "r.-", linewidth=2, label=r"$\hat{y}$")
+
+plt.figure(figsize=(11, 4))
+plt.subplot(121)
+plot_regression_predictions(tree_reg1, X, y)
+for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
+ plt.plot([split, split], [-0.2, 1], style, linewidth=2)
+plt.text(0.21, 0.65, "Depth=0", fontsize=15)
+plt.text(0.01, 0.2, "Depth=1", fontsize=13)
+plt.text(0.65, 0.8, "Depth=1", fontsize=13)
+plt.legend(loc="upper center", fontsize=18)
+plt.title("max_depth=2", fontsize=14)
+
+plt.subplot(122)
+plot_regression_predictions(tree_reg2, X, y, ylabel=None)
+for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
+ plt.plot([split, split], [-0.2, 1], style, linewidth=2)
+for split in (0.0458, 0.1298, 0.2873, 0.9040):
+ plt.plot([split, split], [-0.2, 1], "k:", linewidth=1)
+plt.text(0.3, 0.5, "Depth=2", fontsize=13)
+plt.title("max_depth=3", fontsize=14)
+
+plt.show()
+
+
+
+
+
tree_reg1 = DecisionTreeRegressor(random_state=42)
+tree_reg2 = DecisionTreeRegressor(random_state=42, min_samples_leaf=10)
+tree_reg1.fit(X, y)
+tree_reg2.fit(X, y)
+
+x1 = np.linspace(0, 1, 500).reshape(-1, 1)
+y_pred1 = tree_reg1.predict(x1)
+y_pred2 = tree_reg2.predict(x1)
+
+plt.figure(figsize=(11, 4))
+
+plt.subplot(121)
+plt.plot(X, y, "b.")
+plt.plot(x1, y_pred1, "r.-", linewidth=2, label=r"$\hat{y}$")
+plt.axis([0, 1, -0.2, 1.1])
+plt.xlabel("$x_1$", fontsize=18)
+plt.ylabel("$y$", fontsize=18, rotation=0)
+plt.legend(loc="upper center", fontsize=18)
+plt.title("No restrictions", fontsize=14)
+
+plt.subplot(122)
+plt.plot(X, y, "b.")
+plt.plot(x1, y_pred2, "r.-", linewidth=2, label=r"$\hat{y}$")
+plt.axis([0, 1, -0.2, 1.1])
+plt.xlabel("$x_1$", fontsize=18)
+plt.title("min_samples_leaf={}".format(tree_reg2.min_samples_leaf), fontsize=14)
+
plt.show()
@@ -215,6 +280,11 @@ plt.show()
-Random forests provide an improvement over bagged trees by way of a -small tweak that decorrelates the trees. +
-As in bagging, we build a -number of decision trees on bootstrapped training samples. But when -building these decision trees, each time a split in a tree is -considered, a random sample of \( m \) predictors is chosen as split -candidates from the full set of \( p \) predictors. The split is allowed to -use only one of those \( m \) predictors. - -
-A fresh sample of \( m \) predictors is -taken at each split, and typically we choose - -$$ -m\approx \sqrt{p}. -$$ - -
-In building a random forest, at -each split in the tree, the algorithm is not even allowed to consider -a majority of the available predictors. - -
-The reason for this is rather clever. Suppose that there is one very -strong predictor in the data set, along with a number of other -moderately strong predictors. Then in the collection of bagged -variable importance random forest trees, most or all of the trees will -use this strong predictor in the top split. Consequently, all of the -bagged trees will look quite similar to each other. Hence the -predictions from the bagged trees will be highly correlated. -Unfortunately, averaging many highly correlated quantities does not -lead to as large of a reduction in variance as averaging many -uncorrelated quanti- ties. In particular, this means that bagging will -not lead to a substantial reduction in variance over a single tree in -this setting. - -
diff --git a/doc/pub/DecisionTrees/html/._DecisionTrees-bs029.html b/doc/pub/DecisionTrees/html/._DecisionTrees-bs029.html index c7afc7f08..0ccfacfb4 100644 --- a/doc/pub/DecisionTrees/html/._DecisionTrees-bs029.html +++ b/doc/pub/DecisionTrees/html/._DecisionTrees-bs029.html @@ -67,30 +67,32 @@ Automatically generated HTML file from DocOnce source ('A Classification Tree', 2, None, '___sec12'), ('Growing a classification tree', 2, None, '___sec13'), ('Classification tree, how to split nodes', 2, None, '___sec14'), - ('The CART (Classification and Regression Tree) algorithm', + ('Visualizing the Tree, Classification', 2, None, '___sec15'), + ('Visualizing the Tree, The Moons', 2, None, '___sec16'), + ('Computing the Gini index', 2, None, '___sec17'), + ('Simple Python Code to read in Data', 2, None, '___sec18'), + ('Computing the Gini Factor', 2, None, '___sec19'), + ('Entropy and the ID3 algorithm', 2, None, '___sec20'), + ('Implementing the ID3 Algorithm', 2, None, '___sec21'), + ('Cancer Data again now with Decision Trees and other Methods', 2, None, - '___sec15'), - ('Entropy and the ID3 algorithm', 2, None, '___sec16'), - ('Implementing the ID3 Algorithm', 2, None, '___sec17'), - ('Cancer Data again now with Decision Trees', - 2, - None, - '___sec18'), - ('Another example, the moons again', 2, None, '___sec19'), - ('Playing around with regions', 2, None, '___sec20'), - ('Regression trees', 2, None, '___sec21'), - ('Final regressor code', 2, None, '___sec22'), - ('Pros and cons of trees, pros', 2, None, '___sec23'), - ('Disadvantages', 2, None, '___sec24'), - ('Bagging', 2, None, '___sec25'), - ('Simple example, head or tail', 2, None, '___sec26'), - ('Random forests', 2, None, '___sec27'), - ('A simple scikit-learn example', 2, None, '___sec28'), - ('Please, not the moons again!', 2, None, '___sec29'), - ('Bagging examples', 2, None, '___sec30'), - ('Then random forests', 2, None, '___sec31'), - ('Boosting and more', 2, None, '___sec32')]} + '___sec22'), + ('Another example, the moons again', 2, None, '___sec23'), + ('Playing around with regions', 2, None, '___sec24'), + ('Regression trees', 2, None, '___sec25'), + ('Final regressor code', 2, None, '___sec26'), + ('Pros and cons of trees, pros', 2, None, '___sec27'), + ('Disadvantages', 2, None, '___sec28'), + ('Bagging', 2, None, '___sec29'), + ('More bagging', 2, None, '___sec30'), + ('Simple example, head or tail', 2, None, '___sec31'), + ('Bagging Example', 2, None, '___sec32'), + ('Random forests', 2, None, '___sec33'), + ('A simple scikit-learn example', 2, None, '___sec34'), + ('Please, not the moons again!', 2, None, '___sec35'), + ('Bagging examples', 2, None, '___sec36'), + ('Then random forests', 2, None, '___sec37')]} end of tocinfo --> @@ -143,24 +145,29 @@ MathJax.Hub.Config({
+
from sklearn.ensemble import RandomForestClassifier
-from sklearn.preprocessing import LabelEncoder
-from sklearn.model_selection import cross_validate
-# Data set not specificied
-X = dataset.XXX
-Y = dataset.YYY
-#Instantiate the model with 100 trees and entropy as splitting criteria
-Random_Forest_model = RandomForestClassifier(n_estimators=100,criterion="entropy")
-#Cross validation
-accuracy = cross_validate(Random_Forest_model,X,Y,cv=10)['test_score']
-
@@ -211,6 +217,11 @@ accuracy = cross_validate(Random_Forest_mode
+The plain decision trees suffer from high +variance. This means that if we split the training data into two parts +at random, and fit a decision tree to both halves, the results that we +get could be quite different. In contrast, a procedure with low +variance will yield similar results if applied repeatedly to distinct +data sets; linear regression tends to have low variance, if the ratio +of \( n \) to \( p \) is moderately large. - -
from sklearn.model_selection import train_test_split
-from sklearn.datasets import make_moons
-
-X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
-X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
-from sklearn.ensemble import RandomForestClassifier
-from sklearn.ensemble import VotingClassifier
-from sklearn.linear_model import LogisticRegression
-from sklearn.svm import SVC
-
-log_clf = LogisticRegression(random_state=42)
-rnd_clf = RandomForestClassifier(random_state=42)
-svm_clf = SVC(random_state=42)
-
-voting_clf = VotingClassifier(
- estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
- voting='hard')
-voting_clf.fit(X_train, y_train)
-+Bootstrap aggregation, or just bagging, is a +general-purpose procedure for reducing the variance of a statistical +learning method. - -
from sklearn.metrics import accuracy_score
-
-for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
- clf.fit(X_train, y_train)
- y_pred = clf.predict(X_test)
- print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
-- - -
log_clf = LogisticRegression(random_state=42)
-rnd_clf = RandomForestClassifier(random_state=42)
-svm_clf = SVC(probability=True, random_state=42)
-
-voting_clf = VotingClassifier(
- estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
- voting='soft')
-voting_clf.fit(X_train, y_train)
-- - -
from sklearn.metrics import accuracy_score
-
-for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
- clf.fit(X_train, y_train)
- y_pred = clf.predict(X_test)
- print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
-
@@ -250,6 +218,11 @@ voting_clf.fit(X_train, y_train)
+Bagging typically results in improved accuracy +over prediction using a single tree. Unfortunately, however, it can be +difficult to interpret the resulting model. Recall that one of the +advantages of decision trees is the attractive and easily interpreted +diagram that results. - -
from sklearn.ensemble import BaggingClassifier
-from sklearn.tree import DecisionTreeClassifier
-
-bag_clf = BaggingClassifier(
- DecisionTreeClassifier(random_state=42), n_estimators=500,
- max_samples=100, bootstrap=True, n_jobs=-1, random_state=42)
-bag_clf.fit(X_train, y_train)
-y_pred = bag_clf.predict(X_test)
-+However, when we bag a large number of trees, it is no longer +possible to represent the resulting statistical learning procedure +using a single tree, and it is no longer clear which variables are +most important to the procedure. Thus, bagging improves prediction +accuracy at the expense of interpretability. Although the collection +of bagged trees is much more difficult to interpret than a single +tree, one can obtain an overall summary of the importance of each +predictor using the MSE (for bagging regression trees) or the Gini +index (for bagging classification trees). In the case of bagging +regression trees, we can record the total amount that the MSE is +decreased due to splits over a given predictor, averaged over all \( B \) possible +trees. A large value indicates an important predictor. Similarly, in +the context of bagging classification trees, we can add up the total +amount that the Gini index is decreased by splits over a given +predictor, averaged over all \( B \) trees. - -
from sklearn.metrics import accuracy_score
-print(accuracy_score(y_test, y_pred))
-- - -
tree_clf = DecisionTreeClassifier(random_state=42)
-tree_clf.fit(X_train, y_train)
-y_pred_tree = tree_clf.predict(X_test)
-print(accuracy_score(y_test, y_pred_tree))
-- - -
from matplotlib.colors import ListedColormap
-
-def plot_decision_boundary(clf, X, y, axes=[-1.5, 2.5, -1, 1.5], alpha=0.5, contour=True):
- x1s = np.linspace(axes[0], axes[1], 100)
- x2s = np.linspace(axes[2], axes[3], 100)
- x1, x2 = np.meshgrid(x1s, x2s)
- X_new = np.c_[x1.ravel(), x2.ravel()]
- y_pred = clf.predict(X_new).reshape(x1.shape)
- custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
- plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
- if contour:
- custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
- plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
- plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", alpha=alpha)
- plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", alpha=alpha)
- plt.axis(axes)
- plt.xlabel(r"$x_1$", fontsize=18)
- plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
-plt.figure(figsize=(11,4))
-plt.subplot(121)
-plot_decision_boundary(tree_clf, X, y)
-plt.title("Decision Tree", fontsize=14)
-plt.subplot(122)
-plot_decision_boundary(bag_clf, X, y)
-plt.title("Decision Trees with Bagging", fontsize=14)
-plt.show()
-
@@ -252,6 +227,11 @@ plt.show()
-
@@ -219,7 +226,7 @@ MathJax.Hub.Config({
-
@@ -627,71 +627,200 @@ $$
+
+
+
+
+
+
-The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3.
+The example we will look at is a classical one in many Machine
+Learning applications. Based on various meteorological features, we
+have several so-called attributes which decide whether we at the end
+will do some outdoor activity like skiing, going for a bike ride etc
+etc. The table here contains the feautures outlook, temperature,
+humidity and wind. The target or output is whether we ride
+(True=1) or whether we do something else that day (False=0). The
+attributes for each feature are then sunny, overcast and rain for the
+outlook, hot, cold and mild for temperature, high and normal for
+humidity and weak and strong for wind.
+
+
+The table here summarizes the various attributes and
+
-
+The above functions (gini, entropy and misclassification error) are
+important components of the so-called CART algorithm. We will discuss
+this algorithm below after we have discussed the information gain
+algorithm ID3.
+
+
+In the example here we have converted all our attributes into numerical values \( 0,1,2 \) etc.
+
+
+
+
+
ID3, learns decision trees by constructing
@@ -848,15 +922,216 @@ attributes at each step while growing the tree.
-more text to come here, material presented during lecture Friday Oct 25.
+import re
+import math
+from collections import deque
+
+
+
+
+
+
+
+
+
+class Node(object):
+ def __init__(self):
+ self.value = None
+ self.next = None
+ self.childs = None
+
+
+
+
+class DecisionTree(object):
+ def __init__(self, sample, attributes, labels):
+ self.sample = sample
+ self.attributes = attributes
+ self.labels = labels
+ self.labelCodes = None
+ self.labelCodesCount = None
+ self.initLabelCodes()
+ # print(self.labelCodes)
+ self.root = None
+ self.entropy = self.getEntropy([x for x in range(len(self.labels))])
+
+
+ def initLabelCodes(self):
+ self.labelCodes = []
+ self.labelCodesCount = []
+ for l in self.labels:
+ if l not in self.labelCodes:
+ self.labelCodes.append(l)
+ self.labelCodesCount.append(0)
+ self.labelCodesCount[self.labelCodes.index(l)] += 1
+
+
+ def getLabelCodeId(self, sampleId):
+ return self.labelCodes.index(self.labels[sampleId])
+
+
+ def getAttributeValues(self, sampleIds, attributeId):
+ vals = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in vals:
+ vals.append(val)
+ # print(vals)
+ return vals
+
+
+ def getEntropy(self, sampleIds):
+ entropy = 0
+ labelCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCount[self.getLabelCodeId(sid)] += 1
+ # print("-ge", labelCount)
+ for lv in labelCount:
+ # print(lv)
+ if lv != 0:
+ entropy += -lv/len(sampleIds) * math.log(lv/len(sampleIds), 2)
+ else:
+ entropy += 0
+ return entropy
+
+
+ def getDominantLabel(self, sampleIds):
+ labelCodesCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCodesCount[self.labelCodes.index(self.labels[sid])] += 1
+ return self.labelCodes[labelCodesCount.index(max(labelCodesCount))]
+
+
+ def getInformationGain(self, sampleIds, attributeId):
+ gain = self.getEntropy(sampleIds)
+ attributeVals = []
+ attributeValsCount = []
+ attributeValsIds = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in attributeVals:
+ attributeVals.append(val)
+ attributeValsCount.append(0)
+ attributeValsIds.append([])
+ vid = attributeVals.index(val)
+ attributeValsCount[vid] += 1
+ attributeValsIds[vid].append(sid)
+ # print("-gig", self.attributes[attributeId])
+ for vc, vids in zip(attributeValsCount, attributeValsIds):
+ # print("-gig", vids)
+ gain -= vc/len(sampleIds) * self.getEntropy(vids)
+ return gain
+
+
+ def getAttributeMaxInformationGain(self, sampleIds, attributeIds):
+ attributesEntropy = [0] * len(attributeIds)
+ for i, attId in zip(range(len(attributeIds)), attributeIds):
+ attributesEntropy[i] = self.getInformationGain(sampleIds, attId)
+ maxId = attributeIds[attributesEntropy.index(max(attributesEntropy))]
+ return self.attributes[maxId], maxId
+
+
+ def isSingleLabeled(self, sampleIds):
+ label = self.labels[sampleIds[0]]
+ for sid in sampleIds:
+ if self.labels[sid] != label:
+ return False
+ return True
+
+
+ def getLabel(self, sampleId):
+ return self.labels[sampleId]
+
+
+ def id3(self):
+ sampleIds = [x for x in range(len(self.sample))]
+ attributeIds = [x for x in range(len(self.attributes))]
+ self.root = self.id3Recv(sampleIds, attributeIds, self.root)
+
+
+ def id3Recv(self, sampleIds, attributeIds, root):
+ root = Node() # Initialize current root
+ if self.isSingleLabeled(sampleIds):
+ root.value = self.labels[sampleIds[0]]
+ return root
+ # print(attributeIds)
+ if len(attributeIds) == 0:
+ root.value = self.getDominantLabel(sampleIds)
+ return root
+ bestAttrName, bestAttrId = self.getAttributeMaxInformationGain(
+ sampleIds, attributeIds)
+ # print(bestAttrName)
+ root.value = bestAttrName
+ root.childs = [] # Create list of children
+ for value in self.getAttributeValues(sampleIds, bestAttrId):
+ # print(value)
+ child = Node()
+ child.value = value
+ root.childs.append(child) # Append new child node to current
+ # root
+ childSampleIds = []
+ for sid in sampleIds:
+ if self.sample[sid][bestAttrId] == value:
+ childSampleIds.append(sid)
+ if len(childSampleIds) == 0:
+ child.next = self.getDominantLabel(sampleIds)
+ else:
+ # print(bestAttrName, bestAttrId)
+ # print(attributeIds)
+ if len(attributeIds) > 0 and bestAttrId in attributeIds:
+ toRemove = attributeIds.index(bestAttrId)
+ attributeIds.pop(toRemove)
+ child.next = self.id3Recv(
+ childSampleIds, attributeIds, child.next)
+ return root
+
+
+ def printTree(self):
+ if self.root:
+ roots = deque()
+ roots.append(self.root)
+ while len(roots) > 0:
+ root = roots.popleft()
+ print(root.value)
+ if root.childs:
+ for child in root.childs:
+ print('({})'.format(child.value))
+ roots.append(child.next)
+ elif root.next:
+ print(root.next)
+
+
+def test():
+ f = open('DataFiles/rideclass.csv')
+ attributes = f.readline().split(',')
+ attributes = attributes[1:len(attributes)-1]
+ print(attributes)
+ sample = f.readlines()
+ f.close()
+ for i in range(len(sample)):
+ sample[i] = re.sub('\d+,', '', sample[i])
+ sample[i] = sample[i].strip().split(',')
+ labels = []
+ for s in sample:
+ labels.append(s.pop())
+ # print(sample)
+ # print(labels)
+ decisionTree = DecisionTree(sample, attributes, labels)
+ print("System entropy {}".format(decisionTree.entropy))
+ decisionTree.id3()
+ decisionTree.printTree()
+
+
+if __name__ == '__main__':
+ test()
@@ -906,7 +1181,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
@@ -979,7 +1254,7 @@ plt.show()
@@ -987,7 +1262,7 @@ plt.show()
Xs = np.random.rand(100, 2) - 0.5
ys = (Xs[:, 0] > 0).astype(np.float32) * 2
-angle = np.pi / 4
+angle = np.pi/4
rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
Xsr = Xs.dot(rotation_matrix)
@@ -1008,7 +1283,7 @@ plt.show()
@@ -1031,7 +1306,7 @@ tree_reg.fit(X, y)
@@ -1110,7 +1385,7 @@ plt.show()
The plain decision trees suffer from high
@@ -1158,6 +1433,11 @@ of \( n \) to \( p \) is moderately large.
Bootstrap aggregation, or just bagging, is a
general-purpose procedure for reducing the variance of a statistical
learning method.
+
Bagging typically results in improved accuracy
@@ -1186,7 +1466,7 @@ predictor, averaged over all \( B \) trees.
@@ -1207,7 +1487,59 @@ plt.show()
+
+
+
Random forests provide an improvement over bagged trees by way of a
@@ -1253,7 +1585,7 @@ this setting.
@@ -1272,7 +1604,7 @@ accuracy = cross_validate(Random_Forest_model,X,Y,cv=Please, not the moons again!
+
@@ -1331,7 +1663,7 @@ voting_clf.fit(X_train, y_train)
@@ -1393,7 +1725,7 @@ plt.show()
@@ -1415,12 +1747,6 @@ np.sum(y_pred == y_pred_rf) / len(y_pred)
The CART (Classification and Regression Tree) algorithm
+Visualizing the Tree, Classification
+import os
+from sklearn.datasets import load_breast_cancer
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.model_selection import train_test_split
+from sklearn.metrics import confusion_matrix
+from sklearn.tree import export_graphviz
+
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import pandas as pd
+import numpy as np
+
+
+cancer = load_breast_cancer()
+X = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+print(X)
+y = pd.Categorical.from_codes(cancer.target, cancer.target_names)
+y = pd.get_dummies(y)
+print(y)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/cancer.dot",
+ feature_names=cancer.feature_names,
+ class_names=cancer.target_names,
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png'
+os.system(cmd)
+
Visualizing the Tree, The Moons
+# Common imports
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.datasets import make_moons
+from sklearn.tree import export_graphviz
+from pydot import graph_from_dot_data
+import pandas as pd
+import os
+
+np.random.seed(42)
+X, y = make_moons(n_samples=100, noise=0.25, random_state=53)
+X_train, X_test, y_train, y_test = train_test_split(X,y,random_state=0)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/moons.dot",
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/moons.dot -o DataFiles/moons.png'
+os.system(cmd)
+
Computing the Gini index
+
+
+
+
+
+Day Outlook Temperature Humidity Wind Ride
+ 1 Sunny Hot High Weak 0
+ 2 Sunny Hot High Strong 1
+ 3 Overcast Hot High Weak 1
+ 4 Rain Mild High Weak 1
+ 5 Rain Cool Normal Weak 1
+ 6 Rain Cool Normal Strong 0
+ 7 Overcast Cool Normal Strong 1
+ 8 Sunny Mild High Weak 0
+ 9 Sunny Cool Normal Weak 1
+ 10 Rain Mild Normal Weak 1
+ 11 Sunny Mild Normal Strong 1
+ 12 Overcast Mild High Strong 1
+ 13 Overcast Hot Normal Weak 1
+
+ 14 Rain Mild High Strong 0 Simple Python Code to read in Data
from random import seed
-from random import randrange
-from csv import reader
-
-# Load a CSV file
-def load_csv(filename):
- file = open(filename, "rb")
- lines = reader(file)
- dataset = list(lines)
- return dataset
-
-# Convert string column to float
-def str_column_to_float(dataset, column):
- for row in dataset:
- row[column] = float(row[column].strip())
-
-# Split a dataset into k folds
-def cross_validation_split(dataset, n_folds):
- dataset_split = list()
- dataset_copy = list(dataset)
- fold_size = int(len(dataset) / n_folds)
- for i in range(n_folds):
- fold = list()
- while len(fold) < fold_size:
- index = randrange(len(dataset_copy))
- fold.append(dataset_copy.pop(index))
- dataset_split.append(fold)
- return dataset_split
-
-# Calculate accuracy percentage
-def accuracy_metric(actual, predicted):
- correct = 0
- for i in range(len(actual)):
- if actual[i] == predicted[i]:
- correct += 1
- return correct / float(len(actual)) * 100.0
-
-# Evaluate an algorithm using a cross validation split
-def evaluate_algorithm(dataset, algorithm, n_folds, *args):
- folds = cross_validation_split(dataset, n_folds)
- scores = list()
- for fold in folds:
- train_set = list(folds)
- train_set.remove(fold)
- train_set = sum(train_set, [])
- test_set = list()
- for row in fold:
- row_copy = list(row)
- test_set.append(row_copy)
- row_copy[-1] = None
- predicted = algorithm(train_set, test_set, *args)
- actual = [row[-1] for row in fold]
- accuracy = accuracy_metric(actual, predicted)
- scores.append(accuracy)
- return scores
-
-# Split a dataset based on an attribute and an attribute value
+
# Common imports
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.preprocessing import StandardScaler, OneHotEncoder
+from sklearn.compose import ColumnTransformer
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import os
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+ os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+ os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+ os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+ return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+ return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+ plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("ride.csv"),'r')
+
+# Read the experimental data with Pandas
+from IPython.display import display
+ridedata = pd.read_csv(infile,names = ('Outlook','Temperature','Humidity','Wind','Ride'))
+ridedata = pd.DataFrame(ridedata)
+display(ridedata)
+# Features and targets
+X = ridedata.loc[:, ridedata.columns != 'Ride'].values
+display(X)
+y = ridedata.loc[:, ridedata.columns == 'Ride'].values
+display(y)
+# Categorical variables to one-hot's
+onehotencoder = OneHotEncoder(categories="auto")
+
+X = ColumnTransformer([("", onehotencoder)]).fit_transform(X)
+y.shape
+
+display(X)
+display(y)
+
Computing the Gini Factor
+
+# Split a dataset based on an attribute and an attribute value
def test_split(index, value, dataset):
left, right = list(), list()
for row in dataset:
@@ -720,7 +849,7 @@ The above functions (gini, entropy and misclassification error) are important co
# weight the group score by its relative size
gini += (1.0 - score) * (size / n_instances)
return gini
-
+
# Select the best split point for a dataset
def get_split(dataset):
class_values = list(set(row[-1] for row in dataset))
@@ -729,89 +858,34 @@ The above functions (gini, entropy and misclassification error) are important co
for row in dataset:
groups = test_split(index, row[index], dataset)
gini = gini_index(groups, class_values)
+ print('X%d < %.3f Gini=%.3f' % ((index+1), row[index], gini))
if gini < b_score:
b_index, b_value, b_score, b_groups = index, row[index], gini, groups
return {'index':b_index, 'value':b_value, 'groups':b_groups}
-# Create a terminal node value
-def to_terminal(group):
- outcomes = [row[-1] for row in group]
- return max(set(outcomes), key=outcomes.count)
-
-# Create child splits for a node or make terminal
-def split(node, max_depth, min_size, depth):
- left, right = node['groups']
- del(node['groups'])
- # check for a no split
- if not left or not right:
- node['left'] = node['right'] = to_terminal(left + right)
- return
- # check for max depth
- if depth >= max_depth:
- node['left'], node['right'] = to_terminal(left), to_terminal(right)
- return
- # process left child
- if len(left) <= min_size:
- node['left'] = to_terminal(left)
- else:
- node['left'] = get_split(left)
- split(node['left'], max_depth, min_size, depth+1)
- # process right child
- if len(right) <= min_size:
- node['right'] = to_terminal(right)
- else:
- node['right'] = get_split(right)
- split(node['right'], max_depth, min_size, depth+1)
-
-# Build a decision tree
-def build_tree(train, max_depth, min_size):
- root = get_split(train)
- split(root, max_depth, min_size, 1)
- return root
-
-# Make a prediction with a decision tree
-def predict(node, row):
- if row[node['index']] < node['value']:
- if isinstance(node['left'], dict):
- return predict(node['left'], row)
- else:
- return node['left']
- else:
- if isinstance(node['right'], dict):
- return predict(node['right'], row)
- else:
- return node['right']
-
-# Classification and Regression Tree Algorithm
-def decision_tree(train, test, max_depth, min_size):
- tree = build_tree(train, max_depth, min_size)
- predictions = list()
- for row in test:
- prediction = predict(tree, row)
- predictions.append(prediction)
- return(predictions)
-
-# Test CART
-seed(1)
-# load and prepare data
-filename = 'DataFiles/rideclass.csv'
-dataset = load_csv(filename)
-# convert string attributes to integers
-for i in range(len(dataset[0])):
- str_column_to_float(dataset, i)
-# evaluate algorithm
-n_folds = 5
-max_depth = 5
-min_size = 10
-scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
-print('Scores: %s' % scores)
-print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+dataset = [[0,0,0,0,0],
+ [0,0,0,1,1],
+ [1,0,0,0,1],
+ [2,1,0,0,1],
+ [2,2,1,0,1],
+ [2,2,1,1,0],
+ [1,2,1,1,1],
+ [0,1,0,0,0],
+ [0,2,1,0,1],
+ [2,1,1,0,1],
+ [0,1,1,1,1],
+ [1,1,0,1,1],
+ [1,0,1,0,1],
+ [2,1,0,1,0]]
+
+split = get_split(dataset)
+print('Split: [X%d < %.3f]' % ((split['index']+1), split['value']))
Entropy and the ID3 algorithm
+Entropy and the ID3 algorithm
Implementing the ID3 Algorithm
+Implementing the ID3 Algorithm
Cancer Data again now with Decision Trees
+Cancer Data again now with Decision Trees and other Methods
Another example, the moons again
+Another example, the moons again
Playing around with regions
+Playing around with regions
Regression trees
+Regression trees
Final regressor code
+Final regressor code
Pros and cons of trees, pros
+Pros and cons of trees, pros
Disadvantages
+Disadvantages
Bagging
+Bagging
More bagging
Simple example, head or tail
+Simple example, head or tail
Random forests
+Bagging Example
+from sklearn.model_selection import train_test_split
+from sklearn.datasets import make_moons
+
+X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
+
+from sklearn.ensemble import RandomForestClassifier
+from sklearn.ensemble import VotingClassifier
+from sklearn.linear_model import LogisticRegression
+from sklearn.svm import SVC
+
+log_clf = LogisticRegression(solver="liblinear", random_state=42)
+rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
+svm_clf = SVC(gamma="auto", random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='hard')
+
+voting_clf.fit(X_train, y_train)
+
+from sklearn.metrics import accuracy_score
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+
+log_clf = LogisticRegression(solver="liblinear", random_state=42)
+rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
+svm_clf = SVC(gamma="auto", probability=True, random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='soft')
+voting_clf.fit(X_train, y_train)
+
+from sklearn.metrics import accuracy_score
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+
Random forests
A simple scikit-learn example
+A simple scikit-learn example
Please, not the moons again!
Bagging examples
+Bagging examples
Then random forests
+Then random forests
Boosting and more
-More material to come here.
-
-
@@ -606,71 +608,196 @@ $$
-
+ + +
import os
+from sklearn.datasets import load_breast_cancer
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.model_selection import train_test_split
+from sklearn.metrics import confusion_matrix
+from sklearn.tree import export_graphviz
+
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import pandas as pd
+import numpy as np
+
+
+cancer = load_breast_cancer()
+X = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+print(X)
+y = pd.Categorical.from_codes(cancer.target, cancer.target_names)
+y = pd.get_dummies(y)
+print(y)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/cancer.dot",
+ feature_names=cancer.feature_names,
+ class_names=cancer.target_names,
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png'
+os.system(cmd)
+
+
+
+
+ + +
# Common imports
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.datasets import make_moons
+from sklearn.tree import export_graphviz
+from pydot import graph_from_dot_data
+import pandas as pd
+import os
+
+np.random.seed(42)
+X, y = make_moons(n_samples=100, noise=0.25, random_state=53)
+X_train, X_test, y_train, y_test = train_test_split(X,y,random_state=0)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/moons.dot",
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/moons.dot -o DataFiles/moons.png'
+os.system(cmd)
+
+
+
+
-The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3. +The example we will look at is a classical one in many Machine +Learning applications. Based on various meteorological features, we +have several so-called attributes which decide whether we at the end +will do some outdoor activity like skiing, going for a bike ride etc +etc. The table here contains the feautures outlook, temperature, +humidity and wind. The target or output is whether we ride +(True=1) or whether we do something else that day (False=0). The +attributes for each feature are then sunny, overcast and rain for the +outlook, hot, cold and mild for temperature, high and normal for +humidity and weak and strong for wind. + +
+The table here summarizes the various attributes and +
| Day | Outlook | Temperature | Humidity | Wind | Ride |
|---|---|---|---|---|---|
| 1 | Sunny | Hot | High | Weak | 0 |
| 2 | Sunny | Hot | High | Strong | 1 |
| 3 | Overcast | Hot | High | Weak | 1 |
| 4 | Rain | Mild | High | Weak | 1 |
| 5 | Rain | Cool | Normal | Weak | 1 |
| 6 | Rain | Cool | Normal | Strong | 0 |
| 7 | Overcast | Cool | Normal | Strong | 1 |
| 8 | Sunny | Mild | High | Weak | 0 |
| 9 | Sunny | Cool | Normal | Weak | 1 |
| 10 | Rain | Mild | Normal | Weak | 1 |
| 11 | Sunny | Mild | Normal | Strong | 1 |
| 12 | Overcast | Mild | High | Strong | 1 |
| 13 | Overcast | Hot | Normal | Weak | 1 |
| 14 | Rain | Mild | High | Strong | 0 |
+
+
+
-
from random import seed
-from random import randrange
-from csv import reader
-
-# Load a CSV file
-def load_csv(filename):
- file = open(filename, "rb")
- lines = reader(file)
- dataset = list(lines)
- return dataset
-
-# Convert string column to float
-def str_column_to_float(dataset, column):
- for row in dataset:
- row[column] = float(row[column].strip())
-
-# Split a dataset into k folds
-def cross_validation_split(dataset, n_folds):
- dataset_split = list()
- dataset_copy = list(dataset)
- fold_size = int(len(dataset) / n_folds)
- for i in range(n_folds):
- fold = list()
- while len(fold) < fold_size:
- index = randrange(len(dataset_copy))
- fold.append(dataset_copy.pop(index))
- dataset_split.append(fold)
- return dataset_split
-
-# Calculate accuracy percentage
-def accuracy_metric(actual, predicted):
- correct = 0
- for i in range(len(actual)):
- if actual[i] == predicted[i]:
- correct += 1
- return correct / float(len(actual)) * 100.0
-
-# Evaluate an algorithm using a cross validation split
-def evaluate_algorithm(dataset, algorithm, n_folds, *args):
- folds = cross_validation_split(dataset, n_folds)
- scores = list()
- for fold in folds:
- train_set = list(folds)
- train_set.remove(fold)
- train_set = sum(train_set, [])
- test_set = list()
- for row in fold:
- row_copy = list(row)
- test_set.append(row_copy)
- row_copy[-1] = None
- predicted = algorithm(train_set, test_set, *args)
- actual = [row[-1] for row in fold]
- accuracy = accuracy_metric(actual, predicted)
- scores.append(accuracy)
- return scores
-
-# Split a dataset based on an attribute and an attribute value
+# Common imports
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.preprocessing import StandardScaler, OneHotEncoder
+from sklearn.compose import ColumnTransformer
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import os
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+ os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+ os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+ os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+ return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+ return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+ plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("ride.csv"),'r')
+
+# Read the experimental data with Pandas
+from IPython.display import display
+ridedata = pd.read_csv(infile,names = ('Outlook','Temperature','Humidity','Wind','Ride'))
+ridedata = pd.DataFrame(ridedata)
+display(ridedata)
+# Features and targets
+X = ridedata.loc[:, ridedata.columns != 'Ride'].values
+display(X)
+y = ridedata.loc[:, ridedata.columns == 'Ride'].values
+display(y)
+# Categorical variables to one-hot's
+onehotencoder = OneHotEncoder(categories="auto")
+
+X = ColumnTransformer([("", onehotencoder)]).fit_transform(X)
+y.shape
+
+display(X)
+display(y)
+
+
+
+
+
Computing the Gini Factor
+
+
+The above functions (gini, entropy and misclassification error) are
+important components of the so-called CART algorithm. We will discuss
+this algorithm below after we have discussed the information gain
+algorithm ID3.
+
+
+In the example here we have converted all our attributes into numerical values \( 0,1,2 \) etc.
+
+
+
+
+
# Split a dataset based on an attribute and an attribute value
def test_split(index, value, dataset):
left, right = list(), list()
for row in dataset:
@@ -699,7 +826,7 @@ The above functions (gini, entropy and misclassification error) are important co
# weight the group score by its relative size
gini += (1.0 - score) * (size / n_instances)
return gini
-
+
# Select the best split point for a dataset
def get_split(dataset):
class_values = list(set(row[-1] for row in dataset))
@@ -708,88 +835,33 @@ The above functions (gini, entropy and misclassification error) are important co
for row in dataset:
groups = test_split(index, row[index], dataset)
gini = gini_index(groups, class_values)
+ print('X%d < %.3f Gini=%.3f' % ((index+1), row[index], gini))
if gini < b_score:
b_index, b_value, b_score, b_groups = index, row[index], gini, groups
return {'index':b_index, 'value':b_value, 'groups':b_groups}
-# Create a terminal node value
-def to_terminal(group):
- outcomes = [row[-1] for row in group]
- return max(set(outcomes), key=outcomes.count)
-
-# Create child splits for a node or make terminal
-def split(node, max_depth, min_size, depth):
- left, right = node['groups']
- del(node['groups'])
- # check for a no split
- if not left or not right:
- node['left'] = node['right'] = to_terminal(left + right)
- return
- # check for max depth
- if depth >= max_depth:
- node['left'], node['right'] = to_terminal(left), to_terminal(right)
- return
- # process left child
- if len(left) <= min_size:
- node['left'] = to_terminal(left)
- else:
- node['left'] = get_split(left)
- split(node['left'], max_depth, min_size, depth+1)
- # process right child
- if len(right) <= min_size:
- node['right'] = to_terminal(right)
- else:
- node['right'] = get_split(right)
- split(node['right'], max_depth, min_size, depth+1)
-
-# Build a decision tree
-def build_tree(train, max_depth, min_size):
- root = get_split(train)
- split(root, max_depth, min_size, 1)
- return root
-
-# Make a prediction with a decision tree
-def predict(node, row):
- if row[node['index']] < node['value']:
- if isinstance(node['left'], dict):
- return predict(node['left'], row)
- else:
- return node['left']
- else:
- if isinstance(node['right'], dict):
- return predict(node['right'], row)
- else:
- return node['right']
-
-# Classification and Regression Tree Algorithm
-def decision_tree(train, test, max_depth, min_size):
- tree = build_tree(train, max_depth, min_size)
- predictions = list()
- for row in test:
- prediction = predict(tree, row)
- predictions.append(prediction)
- return(predictions)
-
-# Test CART
-seed(1)
-# load and prepare data
-filename = 'DataFiles/rideclass.csv'
-dataset = load_csv(filename)
-# convert string attributes to integers
-for i in range(len(dataset[0])):
- str_column_to_float(dataset, i)
-# evaluate algorithm
-n_folds = 5
-max_depth = 5
-min_size = 10
-scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
-print('Scores: %s' % scores)
-print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+dataset = [[0,0,0,0,0],
+ [0,0,0,1,1],
+ [1,0,0,0,1],
+ [2,1,0,0,1],
+ [2,2,1,0,1],
+ [2,2,1,1,0],
+ [1,2,1,1,1],
+ [0,1,0,0,0],
+ [0,2,1,0,1],
+ [2,1,1,0,1],
+ [0,1,1,1,1],
+ [1,1,0,1,1],
+ [1,0,1,0,1],
+ [2,1,0,1,0]]
+
+split = get_split(dataset)
+print('Split: [X%d < %.3f]' % ((split['index']+1), split['value']))
-
Entropy and the ID3 algorithm
+Entropy and the ID3 algorithm
ID3, learns decision trees by constructing
@@ -825,15 +897,216 @@ attributes at each step while growing the tree.
-
Implementing the ID3 Algorithm
+Implementing the ID3 Algorithm
-more text to come here, material presented during lecture Friday Oct 25.
+import re
+import math
+from collections import deque
+
+
+
+
+
+
+
+
+
+class Node(object):
+ def __init__(self):
+ self.value = None
+ self.next = None
+ self.childs = None
+
+
+
+
+class DecisionTree(object):
+ def __init__(self, sample, attributes, labels):
+ self.sample = sample
+ self.attributes = attributes
+ self.labels = labels
+ self.labelCodes = None
+ self.labelCodesCount = None
+ self.initLabelCodes()
+ # print(self.labelCodes)
+ self.root = None
+ self.entropy = self.getEntropy([x for x in range(len(self.labels))])
+
+
+ def initLabelCodes(self):
+ self.labelCodes = []
+ self.labelCodesCount = []
+ for l in self.labels:
+ if l not in self.labelCodes:
+ self.labelCodes.append(l)
+ self.labelCodesCount.append(0)
+ self.labelCodesCount[self.labelCodes.index(l)] += 1
+
+
+ def getLabelCodeId(self, sampleId):
+ return self.labelCodes.index(self.labels[sampleId])
+
+
+ def getAttributeValues(self, sampleIds, attributeId):
+ vals = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in vals:
+ vals.append(val)
+ # print(vals)
+ return vals
+
+
+ def getEntropy(self, sampleIds):
+ entropy = 0
+ labelCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCount[self.getLabelCodeId(sid)] += 1
+ # print("-ge", labelCount)
+ for lv in labelCount:
+ # print(lv)
+ if lv != 0:
+ entropy += -lv/len(sampleIds) * math.log(lv/len(sampleIds), 2)
+ else:
+ entropy += 0
+ return entropy
+
+
+ def getDominantLabel(self, sampleIds):
+ labelCodesCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCodesCount[self.labelCodes.index(self.labels[sid])] += 1
+ return self.labelCodes[labelCodesCount.index(max(labelCodesCount))]
+
+
+ def getInformationGain(self, sampleIds, attributeId):
+ gain = self.getEntropy(sampleIds)
+ attributeVals = []
+ attributeValsCount = []
+ attributeValsIds = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in attributeVals:
+ attributeVals.append(val)
+ attributeValsCount.append(0)
+ attributeValsIds.append([])
+ vid = attributeVals.index(val)
+ attributeValsCount[vid] += 1
+ attributeValsIds[vid].append(sid)
+ # print("-gig", self.attributes[attributeId])
+ for vc, vids in zip(attributeValsCount, attributeValsIds):
+ # print("-gig", vids)
+ gain -= vc/len(sampleIds) * self.getEntropy(vids)
+ return gain
+
+
+ def getAttributeMaxInformationGain(self, sampleIds, attributeIds):
+ attributesEntropy = [0] * len(attributeIds)
+ for i, attId in zip(range(len(attributeIds)), attributeIds):
+ attributesEntropy[i] = self.getInformationGain(sampleIds, attId)
+ maxId = attributeIds[attributesEntropy.index(max(attributesEntropy))]
+ return self.attributes[maxId], maxId
+
+
+ def isSingleLabeled(self, sampleIds):
+ label = self.labels[sampleIds[0]]
+ for sid in sampleIds:
+ if self.labels[sid] != label:
+ return False
+ return True
+
+
+ def getLabel(self, sampleId):
+ return self.labels[sampleId]
+
+
+ def id3(self):
+ sampleIds = [x for x in range(len(self.sample))]
+ attributeIds = [x for x in range(len(self.attributes))]
+ self.root = self.id3Recv(sampleIds, attributeIds, self.root)
+
+
+ def id3Recv(self, sampleIds, attributeIds, root):
+ root = Node() # Initialize current root
+ if self.isSingleLabeled(sampleIds):
+ root.value = self.labels[sampleIds[0]]
+ return root
+ # print(attributeIds)
+ if len(attributeIds) == 0:
+ root.value = self.getDominantLabel(sampleIds)
+ return root
+ bestAttrName, bestAttrId = self.getAttributeMaxInformationGain(
+ sampleIds, attributeIds)
+ # print(bestAttrName)
+ root.value = bestAttrName
+ root.childs = [] # Create list of children
+ for value in self.getAttributeValues(sampleIds, bestAttrId):
+ # print(value)
+ child = Node()
+ child.value = value
+ root.childs.append(child) # Append new child node to current
+ # root
+ childSampleIds = []
+ for sid in sampleIds:
+ if self.sample[sid][bestAttrId] == value:
+ childSampleIds.append(sid)
+ if len(childSampleIds) == 0:
+ child.next = self.getDominantLabel(sampleIds)
+ else:
+ # print(bestAttrName, bestAttrId)
+ # print(attributeIds)
+ if len(attributeIds) > 0 and bestAttrId in attributeIds:
+ toRemove = attributeIds.index(bestAttrId)
+ attributeIds.pop(toRemove)
+ child.next = self.id3Recv(
+ childSampleIds, attributeIds, child.next)
+ return root
+
+
+ def printTree(self):
+ if self.root:
+ roots = deque()
+ roots.append(self.root)
+ while len(roots) > 0:
+ root = roots.popleft()
+ print(root.value)
+ if root.childs:
+ for child in root.childs:
+ print('({})'.format(child.value))
+ roots.append(child.next)
+ elif root.next:
+ print(root.next)
+
+
+def test():
+ f = open('DataFiles/rideclass.csv')
+ attributes = f.readline().split(',')
+ attributes = attributes[1:len(attributes)-1]
+ print(attributes)
+ sample = f.readlines()
+ f.close()
+ for i in range(len(sample)):
+ sample[i] = re.sub('\d+,', '', sample[i])
+ sample[i] = sample[i].strip().split(',')
+ labels = []
+ for s in sample:
+ labels.append(s.pop())
+ # print(sample)
+ # print(labels)
+ decisionTree = DecisionTree(sample, attributes, labels)
+ print("System entropy {}".format(decisionTree.entropy))
+ decisionTree.id3()
+ decisionTree.printTree()
+
+
+if __name__ == '__main__':
+ test()
-
Cancer Data again now with Decision Trees
+Cancer Data again now with Decision Trees and other Methods
@@ -882,7 +1155,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
-
Another example, the moons again
+Another example, the moons again
@@ -954,7 +1227,7 @@ plt.show()
-
Playing around with regions
+Playing around with regions
@@ -962,7 +1235,7 @@ plt.show()
Xs = np.random.rand(100, 2) - 0.5
ys = (Xs[:, 0] > 0).astype(np.float32) * 2
-angle = np.pi / 4
+angle = np.pi/4
rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
Xsr = Xs.dot(rotation_matrix)
@@ -982,7 +1255,7 @@ plt.show()
-
Regression trees
+Regression trees
@@ -1004,7 +1277,7 @@ tree_reg.fit(X, y)
-
Final regressor code
+Final regressor code
@@ -1082,7 +1355,7 @@ plt.show()
-
Pros and cons of trees, pros
+Pros and cons of trees, pros
-
The plain decision trees suffer from high @@ -1129,6 +1402,11 @@ of \( n \) to \( p \) is moderately large. general-purpose procedure for reducing the variance of a statistical learning method. +
+
+
+
Bagging typically results in improved accuracy over prediction using a single tree. Unfortunately, however, it can be @@ -1156,7 +1434,7 @@ predictor, averaged over all \( B \) trees.
-
@@ -1176,7 +1454,58 @@ plt.show()
-
+ + +
from sklearn.model_selection import train_test_split
+from sklearn.datasets import make_moons
+
+X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
+
+from sklearn.ensemble import RandomForestClassifier
+from sklearn.ensemble import VotingClassifier
+from sklearn.linear_model import LogisticRegression
+from sklearn.svm import SVC
+
+log_clf = LogisticRegression(solver="liblinear", random_state=42)
+rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
+svm_clf = SVC(gamma="auto", random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='hard')
+
+voting_clf.fit(X_train, y_train)
+
+from sklearn.metrics import accuracy_score
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+
+log_clf = LogisticRegression(solver="liblinear", random_state=42)
+rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
+svm_clf = SVC(gamma="auto", probability=True, random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='soft')
+voting_clf.fit(X_train, y_train)
+
+from sklearn.metrics import accuracy_score
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+
+
+
+
Random forests provide an improvement over bagged trees by way of a @@ -1220,7 +1549,7 @@ this setting.
-
@@ -1238,7 +1567,7 @@ accuracy = cross_validate(Random_Forest_model,X,Y,cv=Please, not the moons again!
+
@@ -1296,7 +1625,7 @@ voting_clf.fit(X_train, y_train)
@@ -1357,7 +1686,7 @@ plt.show()
@@ -1376,12 +1705,6 @@ rnd_clf.fit(X_train, y_train)
y_pred_rf = rnd_clf.predict(X_test)
np.sum(y_pred == y_pred_rf) / len(y_pred)
Please, not the moons again!
-Bagging examples
+Bagging examples
-Then random forests
+Then random forests
- - -
diff --git a/doc/pub/DecisionTrees/html/DecisionTrees.html b/doc/pub/DecisionTrees/html/DecisionTrees.html index dfb71089a..5885a1674 100644 --- a/doc/pub/DecisionTrees/html/DecisionTrees.html +++ b/doc/pub/DecisionTrees/html/DecisionTrees.html @@ -92,30 +92,32 @@ div { text-align: justify; text-justify: inter-word; } ('A Classification Tree', 2, None, '___sec12'), ('Growing a classification tree', 2, None, '___sec13'), ('Classification tree, how to split nodes', 2, None, '___sec14'), - ('The CART (Classification and Regression Tree) algorithm', + ('Visualizing the Tree, Classification', 2, None, '___sec15'), + ('Visualizing the Tree, The Moons', 2, None, '___sec16'), + ('Computing the Gini index', 2, None, '___sec17'), + ('Simple Python Code to read in Data', 2, None, '___sec18'), + ('Computing the Gini Factor', 2, None, '___sec19'), + ('Entropy and the ID3 algorithm', 2, None, '___sec20'), + ('Implementing the ID3 Algorithm', 2, None, '___sec21'), + ('Cancer Data again now with Decision Trees and other Methods', 2, None, - '___sec15'), - ('Entropy and the ID3 algorithm', 2, None, '___sec16'), - ('Implementing the ID3 Algorithm', 2, None, '___sec17'), - ('Cancer Data again now with Decision Trees', - 2, - None, - '___sec18'), - ('Another example, the moons again', 2, None, '___sec19'), - ('Playing around with regions', 2, None, '___sec20'), - ('Regression trees', 2, None, '___sec21'), - ('Final regressor code', 2, None, '___sec22'), - ('Pros and cons of trees, pros', 2, None, '___sec23'), - ('Disadvantages', 2, None, '___sec24'), - ('Bagging', 2, None, '___sec25'), - ('Simple example, head or tail', 2, None, '___sec26'), - ('Random forests', 2, None, '___sec27'), - ('A simple scikit-learn example', 2, None, '___sec28'), - ('Please, not the moons again!', 2, None, '___sec29'), - ('Bagging examples', 2, None, '___sec30'), - ('Then random forests', 2, None, '___sec31'), - ('Boosting and more', 2, None, '___sec32')]} + '___sec22'), + ('Another example, the moons again', 2, None, '___sec23'), + ('Playing around with regions', 2, None, '___sec24'), + ('Regression trees', 2, None, '___sec25'), + ('Final regressor code', 2, None, '___sec26'), + ('Pros and cons of trees, pros', 2, None, '___sec27'), + ('Disadvantages', 2, None, '___sec28'), + ('Bagging', 2, None, '___sec29'), + ('More bagging', 2, None, '___sec30'), + ('Simple example, head or tail', 2, None, '___sec31'), + ('Bagging Example', 2, None, '___sec32'), + ('Random forests', 2, None, '___sec33'), + ('A simple scikit-learn example', 2, None, '___sec34'), + ('Please, not the moons again!', 2, None, '___sec35'), + ('Bagging examples', 2, None, '___sec36'), + ('Then random forests', 2, None, '___sec37')]} end of tocinfo -->
@@ -157,7 +159,7 @@ MathJax.Hub.Config({-
@@ -611,71 +613,196 @@ $$
-
+ + +
import os
+from sklearn.datasets import load_breast_cancer
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.model_selection import train_test_split
+from sklearn.metrics import confusion_matrix
+from sklearn.tree import export_graphviz
+
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import pandas as pd
+import numpy as np
+
+
+cancer = load_breast_cancer()
+X = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+print(X)
+y = pd.Categorical.from_codes(cancer.target, cancer.target_names)
+y = pd.get_dummies(y)
+print(y)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/cancer.dot",
+ feature_names=cancer.feature_names,
+ class_names=cancer.target_names,
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png'
+os.system(cmd)
+
+
+
+
+ + +
# Common imports
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.datasets import make_moons
+from sklearn.tree import export_graphviz
+from pydot import graph_from_dot_data
+import pandas as pd
+import os
+
+np.random.seed(42)
+X, y = make_moons(n_samples=100, noise=0.25, random_state=53)
+X_train, X_test, y_train, y_test = train_test_split(X,y,random_state=0)
+tree_clf = DecisionTreeClassifier(max_depth=5)
+tree_clf.fit(X_train, y_train)
+
+export_graphviz(
+ tree_clf,
+ out_file="DataFiles/moons.dot",
+ rounded=True,
+ filled=True
+)
+cmd = 'dot -Tpng DataFiles/moons.dot -o DataFiles/moons.png'
+os.system(cmd)
+
+
+
+
-The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3. +The example we will look at is a classical one in many Machine +Learning applications. Based on various meteorological features, we +have several so-called attributes which decide whether we at the end +will do some outdoor activity like skiing, going for a bike ride etc +etc. The table here contains the feautures outlook, temperature, +humidity and wind. The target or output is whether we ride +(True=1) or whether we do something else that day (False=0). The +attributes for each feature are then sunny, overcast and rain for the +outlook, hot, cold and mild for temperature, high and normal for +humidity and weak and strong for wind. + +
+The table here summarizes the various attributes and +
| Day | Outlook | Temperature | Humidity | Wind | Ride |
|---|---|---|---|---|---|
| 1 | Sunny | Hot | High | Weak | 0 |
| 2 | Sunny | Hot | High | Strong | 1 |
| 3 | Overcast | Hot | High | Weak | 1 |
| 4 | Rain | Mild | High | Weak | 1 |
| 5 | Rain | Cool | Normal | Weak | 1 |
| 6 | Rain | Cool | Normal | Strong | 0 |
| 7 | Overcast | Cool | Normal | Strong | 1 |
| 8 | Sunny | Mild | High | Weak | 0 |
| 9 | Sunny | Cool | Normal | Weak | 1 |
| 10 | Rain | Mild | Normal | Weak | 1 |
| 11 | Sunny | Mild | Normal | Strong | 1 |
| 12 | Overcast | Mild | High | Strong | 1 |
| 13 | Overcast | Hot | Normal | Weak | 1 |
| 14 | Rain | Mild | High | Strong | 0 |
+
+
+
-
from random import seed
-from random import randrange
-from csv import reader
-
-# Load a CSV file
-def load_csv(filename):
- file = open(filename, "rb")
- lines = reader(file)
- dataset = list(lines)
- return dataset
-
-# Convert string column to float
-def str_column_to_float(dataset, column):
- for row in dataset:
- row[column] = float(row[column].strip())
-
-# Split a dataset into k folds
-def cross_validation_split(dataset, n_folds):
- dataset_split = list()
- dataset_copy = list(dataset)
- fold_size = int(len(dataset) / n_folds)
- for i in range(n_folds):
- fold = list()
- while len(fold) < fold_size:
- index = randrange(len(dataset_copy))
- fold.append(dataset_copy.pop(index))
- dataset_split.append(fold)
- return dataset_split
-
-# Calculate accuracy percentage
-def accuracy_metric(actual, predicted):
- correct = 0
- for i in range(len(actual)):
- if actual[i] == predicted[i]:
- correct += 1
- return correct / float(len(actual)) * 100.0
-
-# Evaluate an algorithm using a cross validation split
-def evaluate_algorithm(dataset, algorithm, n_folds, *args):
- folds = cross_validation_split(dataset, n_folds)
- scores = list()
- for fold in folds:
- train_set = list(folds)
- train_set.remove(fold)
- train_set = sum(train_set, [])
- test_set = list()
- for row in fold:
- row_copy = list(row)
- test_set.append(row_copy)
- row_copy[-1] = None
- predicted = algorithm(train_set, test_set, *args)
- actual = [row[-1] for row in fold]
- accuracy = accuracy_metric(actual, predicted)
- scores.append(accuracy)
- return scores
-
-# Split a dataset based on an attribute and an attribute value
+# Common imports
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.preprocessing import StandardScaler, OneHotEncoder
+from sklearn.compose import ColumnTransformer
+from IPython.display import Image
+from pydot import graph_from_dot_data
+import os
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+ os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+ os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+ os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+ return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+ return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+ plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("ride.csv"),'r')
+
+# Read the experimental data with Pandas
+from IPython.display import display
+ridedata = pd.read_csv(infile,names = ('Outlook','Temperature','Humidity','Wind','Ride'))
+ridedata = pd.DataFrame(ridedata)
+display(ridedata)
+# Features and targets
+X = ridedata.loc[:, ridedata.columns != 'Ride'].values
+display(X)
+y = ridedata.loc[:, ridedata.columns == 'Ride'].values
+display(y)
+# Categorical variables to one-hot's
+onehotencoder = OneHotEncoder(categories="auto")
+
+X = ColumnTransformer([("", onehotencoder)]).fit_transform(X)
+y.shape
+
+display(X)
+display(y)
+
+
+
+
+
Computing the Gini Factor
+
+
+The above functions (gini, entropy and misclassification error) are
+important components of the so-called CART algorithm. We will discuss
+this algorithm below after we have discussed the information gain
+algorithm ID3.
+
+
+In the example here we have converted all our attributes into numerical values \( 0,1,2 \) etc.
+
+
+
+
+
# Split a dataset based on an attribute and an attribute value
def test_split(index, value, dataset):
left, right = list(), list()
for row in dataset:
@@ -704,7 +831,7 @@ The above functions (gini, entropy and misclassification error) are important co
# weight the group score by its relative size
gini += (1.0 - score) * (size / n_instances)
return gini
-
+
# Select the best split point for a dataset
def get_split(dataset):
class_values = list(set(row[-1] for row in dataset))
@@ -713,88 +840,33 @@ The above functions (gini, entropy and misclassification error) are important co
for row in dataset:
groups = test_split(index, row[index], dataset)
gini = gini_index(groups, class_values)
+ print('X%d < %.3f Gini=%.3f' % ((index+1), row[index], gini))
if gini < b_score:
b_index, b_value, b_score, b_groups = index, row[index], gini, groups
return {'index':b_index, 'value':b_value, 'groups':b_groups}
-# Create a terminal node value
-def to_terminal(group):
- outcomes = [row[-1] for row in group]
- return max(set(outcomes), key=outcomes.count)
-
-# Create child splits for a node or make terminal
-def split(node, max_depth, min_size, depth):
- left, right = node['groups']
- del(node['groups'])
- # check for a no split
- if not left or not right:
- node['left'] = node['right'] = to_terminal(left + right)
- return
- # check for max depth
- if depth >= max_depth:
- node['left'], node['right'] = to_terminal(left), to_terminal(right)
- return
- # process left child
- if len(left) <= min_size:
- node['left'] = to_terminal(left)
- else:
- node['left'] = get_split(left)
- split(node['left'], max_depth, min_size, depth+1)
- # process right child
- if len(right) <= min_size:
- node['right'] = to_terminal(right)
- else:
- node['right'] = get_split(right)
- split(node['right'], max_depth, min_size, depth+1)
-
-# Build a decision tree
-def build_tree(train, max_depth, min_size):
- root = get_split(train)
- split(root, max_depth, min_size, 1)
- return root
-
-# Make a prediction with a decision tree
-def predict(node, row):
- if row[node['index']] < node['value']:
- if isinstance(node['left'], dict):
- return predict(node['left'], row)
- else:
- return node['left']
- else:
- if isinstance(node['right'], dict):
- return predict(node['right'], row)
- else:
- return node['right']
-
-# Classification and Regression Tree Algorithm
-def decision_tree(train, test, max_depth, min_size):
- tree = build_tree(train, max_depth, min_size)
- predictions = list()
- for row in test:
- prediction = predict(tree, row)
- predictions.append(prediction)
- return(predictions)
-
-# Test CART
-seed(1)
-# load and prepare data
-filename = 'DataFiles/rideclass.csv'
-dataset = load_csv(filename)
-# convert string attributes to integers
-for i in range(len(dataset[0])):
- str_column_to_float(dataset, i)
-# evaluate algorithm
-n_folds = 5
-max_depth = 5
-min_size = 10
-scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
-print('Scores: %s' % scores)
-print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+dataset = [[0,0,0,0,0],
+ [0,0,0,1,1],
+ [1,0,0,0,1],
+ [2,1,0,0,1],
+ [2,2,1,0,1],
+ [2,2,1,1,0],
+ [1,2,1,1,1],
+ [0,1,0,0,0],
+ [0,2,1,0,1],
+ [2,1,1,0,1],
+ [0,1,1,1,1],
+ [1,1,0,1,1],
+ [1,0,1,0,1],
+ [2,1,0,1,0]]
+
+split = get_split(dataset)
+print('Split: [X%d < %.3f]' % ((split['index']+1), split['value']))
-
Entropy and the ID3 algorithm
+Entropy and the ID3 algorithm
ID3, learns decision trees by constructing
@@ -830,15 +902,216 @@ attributes at each step while growing the tree.
-
Implementing the ID3 Algorithm
+Implementing the ID3 Algorithm
-more text to come here, material presented during lecture Friday Oct 25.
+import re
+import math
+from collections import deque
+
+
+
+
+
+
+
+
+
+class Node(object):
+ def __init__(self):
+ self.value = None
+ self.next = None
+ self.childs = None
+
+
+
+
+class DecisionTree(object):
+ def __init__(self, sample, attributes, labels):
+ self.sample = sample
+ self.attributes = attributes
+ self.labels = labels
+ self.labelCodes = None
+ self.labelCodesCount = None
+ self.initLabelCodes()
+ # print(self.labelCodes)
+ self.root = None
+ self.entropy = self.getEntropy([x for x in range(len(self.labels))])
+
+
+ def initLabelCodes(self):
+ self.labelCodes = []
+ self.labelCodesCount = []
+ for l in self.labels:
+ if l not in self.labelCodes:
+ self.labelCodes.append(l)
+ self.labelCodesCount.append(0)
+ self.labelCodesCount[self.labelCodes.index(l)] += 1
+
+
+ def getLabelCodeId(self, sampleId):
+ return self.labelCodes.index(self.labels[sampleId])
+
+
+ def getAttributeValues(self, sampleIds, attributeId):
+ vals = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in vals:
+ vals.append(val)
+ # print(vals)
+ return vals
+
+
+ def getEntropy(self, sampleIds):
+ entropy = 0
+ labelCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCount[self.getLabelCodeId(sid)] += 1
+ # print("-ge", labelCount)
+ for lv in labelCount:
+ # print(lv)
+ if lv != 0:
+ entropy += -lv/len(sampleIds) * math.log(lv/len(sampleIds), 2)
+ else:
+ entropy += 0
+ return entropy
+
+
+ def getDominantLabel(self, sampleIds):
+ labelCodesCount = [0] * len(self.labelCodes)
+ for sid in sampleIds:
+ labelCodesCount[self.labelCodes.index(self.labels[sid])] += 1
+ return self.labelCodes[labelCodesCount.index(max(labelCodesCount))]
+
+
+ def getInformationGain(self, sampleIds, attributeId):
+ gain = self.getEntropy(sampleIds)
+ attributeVals = []
+ attributeValsCount = []
+ attributeValsIds = []
+ for sid in sampleIds:
+ val = self.sample[sid][attributeId]
+ if val not in attributeVals:
+ attributeVals.append(val)
+ attributeValsCount.append(0)
+ attributeValsIds.append([])
+ vid = attributeVals.index(val)
+ attributeValsCount[vid] += 1
+ attributeValsIds[vid].append(sid)
+ # print("-gig", self.attributes[attributeId])
+ for vc, vids in zip(attributeValsCount, attributeValsIds):
+ # print("-gig", vids)
+ gain -= vc/len(sampleIds) * self.getEntropy(vids)
+ return gain
+
+
+ def getAttributeMaxInformationGain(self, sampleIds, attributeIds):
+ attributesEntropy = [0] * len(attributeIds)
+ for i, attId in zip(range(len(attributeIds)), attributeIds):
+ attributesEntropy[i] = self.getInformationGain(sampleIds, attId)
+ maxId = attributeIds[attributesEntropy.index(max(attributesEntropy))]
+ return self.attributes[maxId], maxId
+
+
+ def isSingleLabeled(self, sampleIds):
+ label = self.labels[sampleIds[0]]
+ for sid in sampleIds:
+ if self.labels[sid] != label:
+ return False
+ return True
+
+
+ def getLabel(self, sampleId):
+ return self.labels[sampleId]
+
+
+ def id3(self):
+ sampleIds = [x for x in range(len(self.sample))]
+ attributeIds = [x for x in range(len(self.attributes))]
+ self.root = self.id3Recv(sampleIds, attributeIds, self.root)
+
+
+ def id3Recv(self, sampleIds, attributeIds, root):
+ root = Node() # Initialize current root
+ if self.isSingleLabeled(sampleIds):
+ root.value = self.labels[sampleIds[0]]
+ return root
+ # print(attributeIds)
+ if len(attributeIds) == 0:
+ root.value = self.getDominantLabel(sampleIds)
+ return root
+ bestAttrName, bestAttrId = self.getAttributeMaxInformationGain(
+ sampleIds, attributeIds)
+ # print(bestAttrName)
+ root.value = bestAttrName
+ root.childs = [] # Create list of children
+ for value in self.getAttributeValues(sampleIds, bestAttrId):
+ # print(value)
+ child = Node()
+ child.value = value
+ root.childs.append(child) # Append new child node to current
+ # root
+ childSampleIds = []
+ for sid in sampleIds:
+ if self.sample[sid][bestAttrId] == value:
+ childSampleIds.append(sid)
+ if len(childSampleIds) == 0:
+ child.next = self.getDominantLabel(sampleIds)
+ else:
+ # print(bestAttrName, bestAttrId)
+ # print(attributeIds)
+ if len(attributeIds) > 0 and bestAttrId in attributeIds:
+ toRemove = attributeIds.index(bestAttrId)
+ attributeIds.pop(toRemove)
+ child.next = self.id3Recv(
+ childSampleIds, attributeIds, child.next)
+ return root
+
+
+ def printTree(self):
+ if self.root:
+ roots = deque()
+ roots.append(self.root)
+ while len(roots) > 0:
+ root = roots.popleft()
+ print(root.value)
+ if root.childs:
+ for child in root.childs:
+ print('({})'.format(child.value))
+ roots.append(child.next)
+ elif root.next:
+ print(root.next)
+
+
+def test():
+ f = open('DataFiles/rideclass.csv')
+ attributes = f.readline().split(',')
+ attributes = attributes[1:len(attributes)-1]
+ print(attributes)
+ sample = f.readlines()
+ f.close()
+ for i in range(len(sample)):
+ sample[i] = re.sub('\d+,', '', sample[i])
+ sample[i] = sample[i].strip().split(',')
+ labels = []
+ for s in sample:
+ labels.append(s.pop())
+ # print(sample)
+ # print(labels)
+ decisionTree = DecisionTree(sample, attributes, labels)
+ print("System entropy {}".format(decisionTree.entropy))
+ decisionTree.id3()
+ decisionTree.printTree()
+
+
+if __name__ == '__main__':
+ test()
-
Cancer Data again now with Decision Trees
+Cancer Data again now with Decision Trees and other Methods
@@ -887,7 +1160,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
-
Another example, the moons again
+Another example, the moons again
@@ -959,7 +1232,7 @@ plt.show()
-
Playing around with regions
+Playing around with regions
@@ -967,7 +1240,7 @@ plt.show()
Xs = np.random.rand(100, 2) - 0.5
ys = (Xs[:, 0] > 0).astype(np.float32) * 2
-angle = np.pi / 4
+angle = np.pi/4
rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
Xsr = Xs.dot(rotation_matrix)
@@ -987,7 +1260,7 @@ plt.show()
-
Regression trees
+Regression trees
@@ -1009,7 +1282,7 @@ tree_reg.fit(X, y)
-
Final regressor code
+Final regressor code
@@ -1087,7 +1360,7 @@ plt.show()
-
Pros and cons of trees, pros
+Pros and cons of trees, pros
-
The plain decision trees suffer from high @@ -1134,6 +1407,11 @@ of \( n \) to \( p \) is moderately large. general-purpose procedure for reducing the variance of a statistical learning method. +
+
+
+
Bagging typically results in improved accuracy over prediction using a single tree. Unfortunately, however, it can be @@ -1161,7 +1439,7 @@ predictor, averaged over all \( B \) trees.
-
@@ -1181,7 +1459,58 @@ plt.show()
-
+ + +
from sklearn.model_selection import train_test_split
+from sklearn.datasets import make_moons
+
+X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
+
+from sklearn.ensemble import RandomForestClassifier
+from sklearn.ensemble import VotingClassifier
+from sklearn.linear_model import LogisticRegression
+from sklearn.svm import SVC
+
+log_clf = LogisticRegression(solver="liblinear", random_state=42)
+rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
+svm_clf = SVC(gamma="auto", random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='hard')
+
+voting_clf.fit(X_train, y_train)
+
+from sklearn.metrics import accuracy_score
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+
+log_clf = LogisticRegression(solver="liblinear", random_state=42)
+rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
+svm_clf = SVC(gamma="auto", probability=True, random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='soft')
+voting_clf.fit(X_train, y_train)
+
+from sklearn.metrics import accuracy_score
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+
+
+
+
Random forests provide an improvement over bagged trees by way of a @@ -1225,7 +1554,7 @@ this setting.
-
@@ -1243,7 +1572,7 @@ accuracy = cross_validate(Random_Forest_mode
-
@@ -1301,7 +1630,7 @@ voting_clf.fit(X_train, y_train)
-
@@ -1362,7 +1691,7 @@ plt.show()
-
@@ -1381,12 +1710,6 @@ rnd_clf.fit(X_train, y_train) y_pred_rf = rnd_clf.predict(X_test) np.sum(y_pred == y_pred_rf) / len(y_pred)
- - -
diff --git a/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb b/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb index ac3afe181..e06ed31ed 100644 --- a/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb +++ b/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb @@ -10,7 +10,7 @@ " \n", "**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University\n", "\n", - "Date: **Oct 29, 2019**\n", + "Date: **Oct 31, 2019**\n", "\n", "Copyright 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license\n", "\n", @@ -506,9 +506,7 @@ "cell_type": "markdown", "metadata": {}, "source": [ - "## The CART (Classification and Regression Tree) algorithm\n", - "\n", - "The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3." + "## Visualizing the Tree, Classification" ] }, { @@ -519,62 +517,210 @@ }, "outputs": [], "source": [ - "from random import seed\n", - "from random import randrange\n", - "from csv import reader\n", - " \n", - "# Load a CSV file\n", - "def load_csv(filename):\n", - "\tfile = open(filename, \"rb\")\n", - "\tlines = reader(file)\n", - "\tdataset = list(lines)\n", - "\treturn dataset\n", - " \n", - "# Convert string column to float\n", - "def str_column_to_float(dataset, column):\n", - "\tfor row in dataset:\n", - "\t\trow[column] = float(row[column].strip())\n", - " \n", - "# Split a dataset into k folds\n", - "def cross_validation_split(dataset, n_folds):\n", - "\tdataset_split = list()\n", - "\tdataset_copy = list(dataset)\n", - "\tfold_size = int(len(dataset) / n_folds)\n", - "\tfor i in range(n_folds):\n", - "\t\tfold = list()\n", - "\t\twhile len(fold) < fold_size:\n", - "\t\t\tindex = randrange(len(dataset_copy))\n", - "\t\t\tfold.append(dataset_copy.pop(index))\n", - "\t\tdataset_split.append(fold)\n", - "\treturn dataset_split\n", - " \n", - "# Calculate accuracy percentage\n", - "def accuracy_metric(actual, predicted):\n", - "\tcorrect = 0\n", - "\tfor i in range(len(actual)):\n", - "\t\tif actual[i] == predicted[i]:\n", - "\t\t\tcorrect += 1\n", - "\treturn correct / float(len(actual)) * 100.0\n", - " \n", - "# Evaluate an algorithm using a cross validation split\n", - "def evaluate_algorithm(dataset, algorithm, n_folds, *args):\n", - "\tfolds = cross_validation_split(dataset, n_folds)\n", - "\tscores = list()\n", - "\tfor fold in folds:\n", - "\t\ttrain_set = list(folds)\n", - "\t\ttrain_set.remove(fold)\n", - "\t\ttrain_set = sum(train_set, [])\n", - "\t\ttest_set = list()\n", - "\t\tfor row in fold:\n", - "\t\t\trow_copy = list(row)\n", - "\t\t\ttest_set.append(row_copy)\n", - "\t\t\trow_copy[-1] = None\n", - "\t\tpredicted = algorithm(train_set, test_set, *args)\n", - "\t\tactual = [row[-1] for row in fold]\n", - "\t\taccuracy = accuracy_metric(actual, predicted)\n", - "\t\tscores.append(accuracy)\n", - "\treturn scores\n", - " \n", + "import os\n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.tree import DecisionTreeClassifier\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.metrics import confusion_matrix\n", + "from sklearn.tree import export_graphviz\n", + "\n", + "from IPython.display import Image \n", + "from pydot import graph_from_dot_data\n", + "import pandas as pd\n", + "import numpy as np\n", + "\n", + "\n", + "cancer = load_breast_cancer()\n", + "X = pd.DataFrame(cancer.data, columns=cancer.feature_names)\n", + "print(X)\n", + "y = pd.Categorical.from_codes(cancer.target, cancer.target_names)\n", + "y = pd.get_dummies(y)\n", + "print(y)\n", + "X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1)\n", + "tree_clf = DecisionTreeClassifier(max_depth=5)\n", + "tree_clf.fit(X_train, y_train)\n", + "\n", + "export_graphviz(\n", + " tree_clf,\n", + " out_file=\"DataFiles/cancer.dot\",\n", + " feature_names=cancer.feature_names,\n", + " class_names=cancer.target_names,\n", + " rounded=True,\n", + " filled=True\n", + ")\n", + "cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png'\n", + "os.system(cmd)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Visualizing the Tree, The Moons" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.tree import DecisionTreeClassifier\n", + "from sklearn.datasets import make_moons\n", + "from sklearn.tree import export_graphviz\n", + "from pydot import graph_from_dot_data\n", + "import pandas as pd\n", + "import os\n", + "\n", + "np.random.seed(42)\n", + "X, y = make_moons(n_samples=100, noise=0.25, random_state=53)\n", + "X_train, X_test, y_train, y_test = train_test_split(X,y,random_state=0)\n", + "tree_clf = DecisionTreeClassifier(max_depth=5)\n", + "tree_clf.fit(X_train, y_train)\n", + "\n", + "export_graphviz(\n", + " tree_clf,\n", + " out_file=\"DataFiles/moons.dot\",\n", + " rounded=True,\n", + " filled=True\n", + ")\n", + "cmd = 'dot -Tpng DataFiles/moons.dot -o DataFiles/moons.png'\n", + "os.system(cmd)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Computing the Gini index\n", + "\n", + "The example we will look at is a classical one in many Machine\n", + "Learning applications. Based on various meteorological features, we\n", + "have several so-called attributes which decide whether we at the end\n", + "will do some outdoor activity like skiing, going for a bike ride etc\n", + "etc. The table here contains the feautures **outlook**, **temperature**,\n", + "**humidity** and **wind**. The target or output is whether we ride\n", + "(True=1) or whether we do something else that day (False=0). The\n", + "attributes for each feature are then sunny, overcast and rain for the\n", + "outlook, hot, cold and mild for temperature, high and normal for\n", + "humidity and weak and strong for wind.\n", + "\n", + "The table here summarizes the various attributes and\n", + "
| Day | Outlook | Temperature | Humidity | Wind | Ride |
|---|---|---|---|---|---|
| 1 | Sunny | Hot | High | Weak | 0 |
| 2 | Sunny | Hot | High | Strong | 1 |
| 3 | Overcast | Hot | High | Weak | 1 |
| 4 | Rain | Mild | High | Weak | 1 |
| 5 | Rain | Cool | Normal | Weak | 1 |
| 6 | Rain | Cool | Normal | Strong | 0 |
| 7 | Overcast | Cool | Normal | Strong | 1 |
| 8 | Sunny | Mild | High | Weak | 0 |
| 9 | Sunny | Cool | Normal | Weak | 1 |
| 10 | Rain | Mild | Normal | Weak | 1 |
| 11 | Sunny | Mild | Normal | Strong | 1 |
| 12 | Overcast | Mild | High | Strong | 1 |
| 13 | Overcast | Hot | Normal | Weak | 1 |
| 14 | Rain | Mild | High | Strong | 0 |