diff --git a/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html b/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html index ceb0e0248..8136086bf 100644 --- a/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html +++ b/doc/pub/DecisionTrees/html/._DecisionTrees-bs000.html @@ -67,26 +67,30 @@ Automatically generated HTML file from DocOnce source ('A Classification Tree', 2, None, '___sec12'), ('Growing a classification tree', 2, None, '___sec13'), ('Classification tree, how to split nodes', 2, None, '___sec14'), - ('Entropy and the ID3 algorithm', 2, None, '___sec15'), - ('Implementing the ID3 Algorithm', 2, None, '___sec16'), + ('The CART (Classification and Regression Tree) algorithm', + 2, + None, + '___sec15'), + ('Entropy and the ID3 algorithm', 2, None, '___sec16'), + ('Implementing the ID3 Algorithm', 2, None, '___sec17'), ('Cancer Data again now with Decision Trees', 2, None, - '___sec17'), - ('Another example, the moons again', 2, None, '___sec18'), - ('Playing around with regions', 2, None, '___sec19'), - ('Regression trees', 2, None, '___sec20'), - ('Final regressor code', 2, None, '___sec21'), - ('Pros and cons of trees, pros', 2, None, '___sec22'), - ('Disadvantages', 2, None, '___sec23'), - ('Bagging', 2, None, '___sec24'), - ('Simple example, head or tail', 2, None, '___sec25'), - ('Random forests', 2, None, '___sec26'), - ('A simple scikit-learn example', 2, None, '___sec27'), - ('Please, not the moons again!', 2, None, '___sec28'), - ('Bagging examples', 2, None, '___sec29'), - ('Then random forests', 2, None, '___sec30'), - ('Boosting and more', 2, None, '___sec31')]} + '___sec18'), + ('Another example, the moons again', 2, None, '___sec19'), + ('Playing around with regions', 2, None, '___sec20'), + ('Regression trees', 2, None, '___sec21'), + ('Final regressor code', 2, None, '___sec22'), + ('Pros and cons of trees, pros', 2, None, '___sec23'), + ('Disadvantages', 2, None, '___sec24'), + ('Bagging', 2, None, '___sec25'), + ('Simple example, head or tail', 2, None, '___sec26'), + ('Random forests', 2, None, '___sec27'), + ('A simple scikit-learn example', 2, None, '___sec28'), + ('Please, not the moons again!', 2, None, '___sec29'), + ('Bagging examples', 2, None, '___sec30'), + ('Then random forests', 2, None, '___sec31'), + ('Boosting and more', 2, None, '___sec32')]} end of tocinfo -->
@@ -139,23 +143,24 @@ MathJax.Hub.Config({-
@@ -214,7 +219,7 @@ MathJax.Hub.Config({
-ID3, learns decision trees by constructing -them topdown, beginning with the question which attribute should be tested at the root of the tree? - -
-We would like to select the attribute that is most useful for classifying -examples. - -
-What is a good quantitative measure of the worth of an attribute? - -
-Information gain measures how well a given attribute separates the -training examples according to their target classification. - -
-The ID3 algorithm uses this information gain measure to select among the candidate -attributes at each step while growing the tree. + +
from random import seed
+from random import randrange
+from csv import reader
+
+# Load a CSV file
+def load_csv(filename):
+ file = open(filename, "rb")
+ lines = reader(file)
+ dataset = list(lines)
+ return dataset
+
+# Convert string column to float
+def str_column_to_float(dataset, column):
+ for row in dataset:
+ row[column] = float(row[column].strip())
+
+# Split a dataset into k folds
+def cross_validation_split(dataset, n_folds):
+ dataset_split = list()
+ dataset_copy = list(dataset)
+ fold_size = int(len(dataset) / n_folds)
+ for i in range(n_folds):
+ fold = list()
+ while len(fold) < fold_size:
+ index = randrange(len(dataset_copy))
+ fold.append(dataset_copy.pop(index))
+ dataset_split.append(fold)
+ return dataset_split
+
+# Calculate accuracy percentage
+def accuracy_metric(actual, predicted):
+ correct = 0
+ for i in range(len(actual)):
+ if actual[i] == predicted[i]:
+ correct += 1
+ return correct / float(len(actual)) * 100.0
+
+# Evaluate an algorithm using a cross validation split
+def evaluate_algorithm(dataset, algorithm, n_folds, *args):
+ folds = cross_validation_split(dataset, n_folds)
+ scores = list()
+ for fold in folds:
+ train_set = list(folds)
+ train_set.remove(fold)
+ train_set = sum(train_set, [])
+ test_set = list()
+ for row in fold:
+ row_copy = list(row)
+ test_set.append(row_copy)
+ row_copy[-1] = None
+ predicted = algorithm(train_set, test_set, *args)
+ actual = [row[-1] for row in fold]
+ accuracy = accuracy_metric(actual, predicted)
+ scores.append(accuracy)
+ return scores
+
+# Split a dataset based on an attribute and an attribute value
+def test_split(index, value, dataset):
+ left, right = list(), list()
+ for row in dataset:
+ if row[index] < value:
+ left.append(row)
+ else:
+ right.append(row)
+ return left, right
+
+# Calculate the Gini index for a split dataset
+def gini_index(groups, classes):
+ # count all samples at split point
+ n_instances = float(sum([len(group) for group in groups]))
+ # sum weighted Gini index for each group
+ gini = 0.0
+ for group in groups:
+ size = float(len(group))
+ # avoid divide by zero
+ if size == 0:
+ continue
+ score = 0.0
+ # score the group based on the score for each class
+ for class_val in classes:
+ p = [row[-1] for row in group].count(class_val) / size
+ score += p * p
+ # weight the group score by its relative size
+ gini += (1.0 - score) * (size / n_instances)
+ return gini
+
+# Select the best split point for a dataset
+def get_split(dataset):
+ class_values = list(set(row[-1] for row in dataset))
+ b_index, b_value, b_score, b_groups = 999, 999, 999, None
+ for index in range(len(dataset[0])-1):
+ for row in dataset:
+ groups = test_split(index, row[index], dataset)
+ gini = gini_index(groups, class_values)
+ if gini < b_score:
+ b_index, b_value, b_score, b_groups = index, row[index], gini, groups
+ return {'index':b_index, 'value':b_value, 'groups':b_groups}
+
+# Create a terminal node value
+def to_terminal(group):
+ outcomes = [row[-1] for row in group]
+ return max(set(outcomes), key=outcomes.count)
+
+# Create child splits for a node or make terminal
+def split(node, max_depth, min_size, depth):
+ left, right = node['groups']
+ del(node['groups'])
+ # check for a no split
+ if not left or not right:
+ node['left'] = node['right'] = to_terminal(left + right)
+ return
+ # check for max depth
+ if depth >= max_depth:
+ node['left'], node['right'] = to_terminal(left), to_terminal(right)
+ return
+ # process left child
+ if len(left) <= min_size:
+ node['left'] = to_terminal(left)
+ else:
+ node['left'] = get_split(left)
+ split(node['left'], max_depth, min_size, depth+1)
+ # process right child
+ if len(right) <= min_size:
+ node['right'] = to_terminal(right)
+ else:
+ node['right'] = get_split(right)
+ split(node['right'], max_depth, min_size, depth+1)
+
+# Build a decision tree
+def build_tree(train, max_depth, min_size):
+ root = get_split(train)
+ split(root, max_depth, min_size, 1)
+ return root
+
+# Make a prediction with a decision tree
+def predict(node, row):
+ if row[node['index']] < node['value']:
+ if isinstance(node['left'], dict):
+ return predict(node['left'], row)
+ else:
+ return node['left']
+ else:
+ if isinstance(node['right'], dict):
+ return predict(node['right'], row)
+ else:
+ return node['right']
+
+# Classification and Regression Tree Algorithm
+def decision_tree(train, test, max_depth, min_size):
+ tree = build_tree(train, max_depth, min_size)
+ predictions = list()
+ for row in test:
+ prediction = predict(tree, row)
+ predictions.append(prediction)
+ return(predictions)
+
+# Test CART
+seed(1)
+# load and prepare data
+filename = 'DataFiles/rideclass.csv'
+dataset = load_csv(filename)
+# convert string attributes to integers
+for i in range(len(dataset[0])):
+ str_column_to_float(dataset, i)
+# evaluate algorithm
+n_folds = 5
+max_depth = 5
+min_size = 10
+scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
+print('Scores: %s' % scores)
+print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+
@@ -230,7 +382,7 @@ attributes at each step while growing the tree.
-more text to come here, material presented during lecture Friday Oct 25. +ID3, learns decision trees by constructing +them topdown, beginning with the question which attribute should be tested at the root of the tree? + +
+We would like to select the attribute that is most useful for classifying +examples. + +
+What is a good quantitative measure of the worth of an attribute? + +
+Information gain measures how well a given attribute separates the +training examples according to their target classification. + +
+The ID3 algorithm uses this information gain measure to select among the candidate +attributes at each step while growing the tree.
@@ -202,7 +235,7 @@ MathJax.Hub.Config({
+more text to come here, material presented during lecture Friday Oct 25. - -
import matplotlib.pyplot as plt
-import numpy as np
-from sklearn.model_selection import train_test_split
-from sklearn.datasets import load_breast_cancer
-from sklearn.svm import SVC
-from sklearn.linear_model import LogisticRegression
-from sklearn.tree import DecisionTreeClassifier
-
-# Load the data
-cancer = load_breast_cancer()
-
-X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
-print(X_train.shape)
-print(X_test.shape)
-# Logistic Regression
-logreg = LogisticRegression(solver='lbfgs')
-logreg.fit(X_train, y_train)
-print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
-# Support vector machine
-svm = SVC(gamma='auto', C=100)
-svm.fit(X_train, y_train)
-print("Test set accuracy with SVM: {:.2f}".format(svm.score(X_test,y_test)))
-# Decision Trees
-deep_tree_clf = DecisionTreeClassifier(max_depth=None)
-deep_tree_clf.fit(X_train, y_train)
-print("Test set accuracy with Decision Trees: {:.2f}".format(deep_tree_clf.score(X_test,y_test)))
-#now scale the data
-from sklearn.preprocessing import StandardScaler
-scaler = StandardScaler()
-scaler.fit(X_train)
-X_train_scaled = scaler.transform(X_train)
-X_test_scaled = scaler.transform(X_test)
-# Logistic Regression
-logreg.fit(X_train_scaled, y_train)
-print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-# Support Vector Machine
-svm.fit(X_train_scaled, y_train)
-print("Test set accuracy SVM with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-# Decision Trees
-deep_tree_clf.fit(X_train_scaled, y_train)
-print("Test set accuracy with Decision Trees and scaled data: {:.2f}".format(deep_tree_clf.score(X_test_scaled,y_test)))
-
@@ -243,7 +207,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
-
from __future__ import division, print_function, unicode_literals
-
-# Common imports
+import matplotlib.pyplot as plt
import numpy as np
-import os
-
-# to make this notebook's output stable across runs
-np.random.seed(42)
-
-# To plot pretty figures
-import matplotlib
-import matplotlib.pyplot as plt
-from matplotlib.colors import ListedColormap
-plt.rcParams['axes.labelsize'] = 14
-plt.rcParams['xtick.labelsize'] = 12
-plt.rcParams['ytick.labelsize'] = 12
-
-
+from sklearn.model_selection import train_test_split
+from sklearn.datasets import load_breast_cancer
from sklearn.svm import SVC
-from sklearn import datasets
+from sklearn.linear_model import LogisticRegression
from sklearn.tree import DecisionTreeClassifier
-from sklearn.datasets import make_moons
-from sklearn.tree import export_graphviz
-Xm, ym = make_moons(n_samples=100, noise=0.25, random_state=53)
+# Load the data
+cancer = load_breast_cancer()
-deep_tree_clf1 = DecisionTreeClassifier(random_state=42)
-deep_tree_clf2 = DecisionTreeClassifier(min_samples_leaf=4, random_state=42)
-deep_tree_clf1.fit(Xm, ym)
-deep_tree_clf2.fit(Xm, ym)
-
-
-def plot_decision_boundary(clf, X, y, axes=[0, 7.5, 0, 3], iris=True, legend=False, plot_training=True):
- x1s = np.linspace(axes[0], axes[1], 100)
- x2s = np.linspace(axes[2], axes[3], 100)
- x1, x2 = np.meshgrid(x1s, x2s)
- X_new = np.c_[x1.ravel(), x2.ravel()]
- y_pred = clf.predict(X_new).reshape(x1.shape)
- custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
- plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
- if not iris:
- custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
- plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
- if plot_training:
- plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", label="Iris-Setosa")
- plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", label="Iris-Versicolor")
- plt.plot(X[:, 0][y==2], X[:, 1][y==2], "g^", label="Iris-Virginica")
- plt.axis(axes)
- if iris:
- plt.xlabel("Petal length", fontsize=14)
- plt.ylabel("Petal width", fontsize=14)
- else:
- plt.xlabel(r"$x_1$", fontsize=18)
- plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
- if legend:
- plt.legend(loc="lower right", fontsize=14)
-plt.figure(figsize=(11, 4))
-plt.subplot(121)
-plot_decision_boundary(deep_tree_clf1, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
-plt.title("No restrictions", fontsize=16)
-plt.subplot(122)
-plot_decision_boundary(deep_tree_clf2, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
-plt.title("min_samples_leaf = {}".format(deep_tree_clf2.min_samples_leaf), fontsize=14)
-plt.show()
+X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
+print(X_train.shape)
+print(X_test.shape)
+# Logistic Regression
+logreg = LogisticRegression(solver='lbfgs')
+logreg.fit(X_train, y_train)
+print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
+# Support vector machine
+svm = SVC(gamma='auto', C=100)
+svm.fit(X_train, y_train)
+print("Test set accuracy with SVM: {:.2f}".format(svm.score(X_test,y_test)))
+# Decision Trees
+deep_tree_clf = DecisionTreeClassifier(max_depth=None)
+deep_tree_clf.fit(X_train, y_train)
+print("Test set accuracy with Decision Trees: {:.2f}".format(deep_tree_clf.score(X_test,y_test)))
+#now scale the data
+from sklearn.preprocessing import StandardScaler
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+# Logistic Regression
+logreg.fit(X_train_scaled, y_train)
+print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+# Support Vector Machine
+svm.fit(X_train_scaled, y_train)
+print("Test set accuracy SVM with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+# Decision Trees
+deep_tree_clf.fit(X_train_scaled, y_train)
+print("Test set accuracy with Decision Trees and scaled data: {:.2f}".format(deep_tree_clf.score(X_test_scaled,y_test)))
@@ -266,7 +248,7 @@ plt.show()
-
np.random.seed(6)
-Xs = np.random.rand(100, 2) - 0.5
-ys = (Xs[:, 0] > 0).astype(np.float32) * 2
+from __future__ import division, print_function, unicode_literals
-angle = np.pi / 4
-rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
-Xsr = Xs.dot(rotation_matrix)
+# Common imports
+import numpy as np
+import os
-tree_clf_s = DecisionTreeClassifier(random_state=42)
-tree_clf_s.fit(Xs, ys)
-tree_clf_sr = DecisionTreeClassifier(random_state=42)
-tree_clf_sr.fit(Xsr, ys)
+# to make this notebook's output stable across runs
+np.random.seed(42)
+# To plot pretty figures
+import matplotlib
+import matplotlib.pyplot as plt
+from matplotlib.colors import ListedColormap
+plt.rcParams['axes.labelsize'] = 14
+plt.rcParams['xtick.labelsize'] = 12
+plt.rcParams['ytick.labelsize'] = 12
+
+
+from sklearn.svm import SVC
+from sklearn import datasets
+from sklearn.tree import DecisionTreeClassifier
+from sklearn.datasets import make_moons
+from sklearn.tree import export_graphviz
+
+Xm, ym = make_moons(n_samples=100, noise=0.25, random_state=53)
+
+deep_tree_clf1 = DecisionTreeClassifier(random_state=42)
+deep_tree_clf2 = DecisionTreeClassifier(min_samples_leaf=4, random_state=42)
+deep_tree_clf1.fit(Xm, ym)
+deep_tree_clf2.fit(Xm, ym)
+
+
+def plot_decision_boundary(clf, X, y, axes=[0, 7.5, 0, 3], iris=True, legend=False, plot_training=True):
+ x1s = np.linspace(axes[0], axes[1], 100)
+ x2s = np.linspace(axes[2], axes[3], 100)
+ x1, x2 = np.meshgrid(x1s, x2s)
+ X_new = np.c_[x1.ravel(), x2.ravel()]
+ y_pred = clf.predict(X_new).reshape(x1.shape)
+ custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
+ plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
+ if not iris:
+ custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
+ plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
+ if plot_training:
+ plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", label="Iris-Setosa")
+ plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", label="Iris-Versicolor")
+ plt.plot(X[:, 0][y==2], X[:, 1][y==2], "g^", label="Iris-Virginica")
+ plt.axis(axes)
+ if iris:
+ plt.xlabel("Petal length", fontsize=14)
+ plt.ylabel("Petal width", fontsize=14)
+ else:
+ plt.xlabel(r"$x_1$", fontsize=18)
+ plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
+ if legend:
+ plt.legend(loc="lower right", fontsize=14)
plt.figure(figsize=(11, 4))
plt.subplot(121)
-plot_decision_boundary(tree_clf_s, Xs, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
+plot_decision_boundary(deep_tree_clf1, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
+plt.title("No restrictions", fontsize=16)
plt.subplot(122)
-plot_decision_boundary(tree_clf_sr, Xsr, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
-
+plot_decision_boundary(deep_tree_clf2, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False)
+plt.title("min_samples_leaf = {}".format(deep_tree_clf2.min_samples_leaf), fontsize=14)
plt.show()
@@ -222,7 +271,7 @@ plt.show()
-
# Quadratic training set + noise
-np.random.seed(42)
-m = 200
-X = np.random.rand(m, 1)
-y = 4 * (X - 0.5) ** 2
-y = y + np.random.randn(m, 1) / 10
-+
np.random.seed(6)
+Xs = np.random.rand(100, 2) - 0.5
+ys = (Xs[:, 0] > 0).astype(np.float32) * 2
-
-from sklearn.tree import DecisionTreeRegressor
+angle = np.pi / 4
+rotation_matrix = np.array([[np.cos(angle), -np.sin(angle)], [np.sin(angle), np.cos(angle)]])
+Xsr = Xs.dot(rotation_matrix)
-tree_reg = DecisionTreeRegressor(max_depth=2, random_state=42)
-tree_reg.fit(X, y)
+tree_clf_s = DecisionTreeClassifier(random_state=42)
+tree_clf_s.fit(Xs, ys)
+tree_clf_sr = DecisionTreeClassifier(random_state=42)
+tree_clf_sr.fit(Xsr, ys)
+
+plt.figure(figsize=(11, 4))
+plt.subplot(121)
+plot_decision_boundary(tree_clf_s, Xs, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
+plt.subplot(122)
+plot_decision_boundary(tree_clf_sr, Xsr, ys, axes=[-0.7, 0.7, -0.7, 0.7], iris=False)
+
+plt.show()
@@ -216,7 +227,7 @@ tree_reg.fit(X, y)
+ + +
# Quadratic training set + noise
+np.random.seed(42)
+m = 200
+X = np.random.rand(m, 1)
+y = 4 * (X - 0.5) ** 2
+y = y + np.random.randn(m, 1) / 10
+
from sklearn.tree import DecisionTreeRegressor
-tree_reg1 = DecisionTreeRegressor(random_state=42, max_depth=2)
-tree_reg2 = DecisionTreeRegressor(random_state=42, max_depth=3)
-tree_reg1.fit(X, y)
-tree_reg2.fit(X, y)
-
-def plot_regression_predictions(tree_reg, X, y, axes=[0, 1, -0.2, 1], ylabel="$y$"):
- x1 = np.linspace(axes[0], axes[1], 500).reshape(-1, 1)
- y_pred = tree_reg.predict(x1)
- plt.axis(axes)
- plt.xlabel("$x_1$", fontsize=18)
- if ylabel:
- plt.ylabel(ylabel, fontsize=18, rotation=0)
- plt.plot(X, y, "b.")
- plt.plot(x1, y_pred, "r.-", linewidth=2, label=r"$\hat{y}$")
-
-plt.figure(figsize=(11, 4))
-plt.subplot(121)
-plot_regression_predictions(tree_reg1, X, y)
-for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
- plt.plot([split, split], [-0.2, 1], style, linewidth=2)
-plt.text(0.21, 0.65, "Depth=0", fontsize=15)
-plt.text(0.01, 0.2, "Depth=1", fontsize=13)
-plt.text(0.65, 0.8, "Depth=1", fontsize=13)
-plt.legend(loc="upper center", fontsize=18)
-plt.title("max_depth=2", fontsize=14)
-
-plt.subplot(122)
-plot_regression_predictions(tree_reg2, X, y, ylabel=None)
-for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
- plt.plot([split, split], [-0.2, 1], style, linewidth=2)
-for split in (0.0458, 0.1298, 0.2873, 0.9040):
- plt.plot([split, split], [-0.2, 1], "k:", linewidth=1)
-plt.text(0.3, 0.5, "Depth=2", fontsize=13)
-plt.title("max_depth=3", fontsize=14)
-
-plt.show()
-- - -
tree_reg1 = DecisionTreeRegressor(random_state=42)
-tree_reg2 = DecisionTreeRegressor(random_state=42, min_samples_leaf=10)
-tree_reg1.fit(X, y)
-tree_reg2.fit(X, y)
-
-x1 = np.linspace(0, 1, 500).reshape(-1, 1)
-y_pred1 = tree_reg1.predict(x1)
-y_pred2 = tree_reg2.predict(x1)
-
-plt.figure(figsize=(11, 4))
-
-plt.subplot(121)
-plt.plot(X, y, "b.")
-plt.plot(x1, y_pred1, "r.-", linewidth=2, label=r"$\hat{y}$")
-plt.axis([0, 1, -0.2, 1.1])
-plt.xlabel("$x_1$", fontsize=18)
-plt.ylabel("$y$", fontsize=18, rotation=0)
-plt.legend(loc="upper center", fontsize=18)
-plt.title("No restrictions", fontsize=14)
-
-plt.subplot(122)
-plt.plot(X, y, "b.")
-plt.plot(x1, y_pred2, "r.-", linewidth=2, label=r"$\hat{y}$")
-plt.axis([0, 1, -0.2, 1.1])
-plt.xlabel("$x_1$", fontsize=18)
-plt.title("min_samples_leaf={}".format(tree_reg2.min_samples_leaf), fontsize=14)
-
-plt.show()
+tree_reg = DecisionTreeRegressor(max_depth=2, random_state=42)
+tree_reg.fit(X, y)
@@ -272,7 +221,7 @@ plt.show()
-
from sklearn.tree import DecisionTreeRegressor
+tree_reg1 = DecisionTreeRegressor(random_state=42, max_depth=2)
+tree_reg2 = DecisionTreeRegressor(random_state=42, max_depth=3)
+tree_reg1.fit(X, y)
+tree_reg2.fit(X, y)
+
+def plot_regression_predictions(tree_reg, X, y, axes=[0, 1, -0.2, 1], ylabel="$y$"):
+ x1 = np.linspace(axes[0], axes[1], 500).reshape(-1, 1)
+ y_pred = tree_reg.predict(x1)
+ plt.axis(axes)
+ plt.xlabel("$x_1$", fontsize=18)
+ if ylabel:
+ plt.ylabel(ylabel, fontsize=18, rotation=0)
+ plt.plot(X, y, "b.")
+ plt.plot(x1, y_pred, "r.-", linewidth=2, label=r"$\hat{y}$")
+
+plt.figure(figsize=(11, 4))
+plt.subplot(121)
+plot_regression_predictions(tree_reg1, X, y)
+for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
+ plt.plot([split, split], [-0.2, 1], style, linewidth=2)
+plt.text(0.21, 0.65, "Depth=0", fontsize=15)
+plt.text(0.01, 0.2, "Depth=1", fontsize=13)
+plt.text(0.65, 0.8, "Depth=1", fontsize=13)
+plt.legend(loc="upper center", fontsize=18)
+plt.title("max_depth=2", fontsize=14)
+
+plt.subplot(122)
+plot_regression_predictions(tree_reg2, X, y, ylabel=None)
+for split, style in ((0.1973, "k-"), (0.0917, "k--"), (0.7718, "k--")):
+ plt.plot([split, split], [-0.2, 1], style, linewidth=2)
+for split in (0.0458, 0.1298, 0.2873, 0.9040):
+ plt.plot([split, split], [-0.2, 1], "k:", linewidth=1)
+plt.text(0.3, 0.5, "Depth=2", fontsize=13)
+plt.title("max_depth=3", fontsize=14)
+
+plt.show()
++ + +
tree_reg1 = DecisionTreeRegressor(random_state=42)
+tree_reg2 = DecisionTreeRegressor(random_state=42, min_samples_leaf=10)
+tree_reg1.fit(X, y)
+tree_reg2.fit(X, y)
+
+x1 = np.linspace(0, 1, 500).reshape(-1, 1)
+y_pred1 = tree_reg1.predict(x1)
+y_pred2 = tree_reg2.predict(x1)
+
+plt.figure(figsize=(11, 4))
+
+plt.subplot(121)
+plt.plot(X, y, "b.")
+plt.plot(x1, y_pred1, "r.-", linewidth=2, label=r"$\hat{y}$")
+plt.axis([0, 1, -0.2, 1.1])
+plt.xlabel("$x_1$", fontsize=18)
+plt.ylabel("$y$", fontsize=18, rotation=0)
+plt.legend(loc="upper center", fontsize=18)
+plt.title("No restrictions", fontsize=14)
+
+plt.subplot(122)
+plt.plot(X, y, "b.")
+plt.plot(x1, y_pred2, "r.-", linewidth=2, label=r"$\hat{y}$")
+plt.axis([0, 1, -0.2, 1.1])
+plt.xlabel("$x_1$", fontsize=18)
+plt.title("min_samples_leaf={}".format(tree_reg2.min_samples_leaf), fontsize=14)
+
+plt.show()
+
diff --git a/doc/pub/DecisionTrees/html/._DecisionTrees-bs024.html b/doc/pub/DecisionTrees/html/._DecisionTrees-bs024.html index 86deef56b..245bac28b 100644 --- a/doc/pub/DecisionTrees/html/._DecisionTrees-bs024.html +++ b/doc/pub/DecisionTrees/html/._DecisionTrees-bs024.html @@ -67,26 +67,30 @@ Automatically generated HTML file from DocOnce source ('A Classification Tree', 2, None, '___sec12'), ('Growing a classification tree', 2, None, '___sec13'), ('Classification tree, how to split nodes', 2, None, '___sec14'), - ('Entropy and the ID3 algorithm', 2, None, '___sec15'), - ('Implementing the ID3 Algorithm', 2, None, '___sec16'), + ('The CART (Classification and Regression Tree) algorithm', + 2, + None, + '___sec15'), + ('Entropy and the ID3 algorithm', 2, None, '___sec16'), + ('Implementing the ID3 Algorithm', 2, None, '___sec17'), ('Cancer Data again now with Decision Trees', 2, None, - '___sec17'), - ('Another example, the moons again', 2, None, '___sec18'), - ('Playing around with regions', 2, None, '___sec19'), - ('Regression trees', 2, None, '___sec20'), - ('Final regressor code', 2, None, '___sec21'), - ('Pros and cons of trees, pros', 2, None, '___sec22'), - ('Disadvantages', 2, None, '___sec23'), - ('Bagging', 2, None, '___sec24'), - ('Simple example, head or tail', 2, None, '___sec25'), - ('Random forests', 2, None, '___sec26'), - ('A simple scikit-learn example', 2, None, '___sec27'), - ('Please, not the moons again!', 2, None, '___sec28'), - ('Bagging examples', 2, None, '___sec29'), - ('Then random forests', 2, None, '___sec30'), - ('Boosting and more', 2, None, '___sec31')]} + '___sec18'), + ('Another example, the moons again', 2, None, '___sec19'), + ('Playing around with regions', 2, None, '___sec20'), + ('Regression trees', 2, None, '___sec21'), + ('Final regressor code', 2, None, '___sec22'), + ('Pros and cons of trees, pros', 2, None, '___sec23'), + ('Disadvantages', 2, None, '___sec24'), + ('Bagging', 2, None, '___sec25'), + ('Simple example, head or tail', 2, None, '___sec26'), + ('Random forests', 2, None, '___sec27'), + ('A simple scikit-learn example', 2, None, '___sec28'), + ('Please, not the moons again!', 2, None, '___sec29'), + ('Bagging examples', 2, None, '___sec30'), + ('Then random forests', 2, None, '___sec31'), + ('Boosting and more', 2, None, '___sec32')]} end of tocinfo --> @@ -139,23 +143,24 @@ MathJax.Hub.Config({
-The plain decision trees suffer from high -variance. This means that if we split the training data into two parts -at random, and fit a decision tree to both halves, the results that we -get could be quite different. In contrast, a procedure with low -variance will yield similar results if applied repeatedly to distinct -data sets; linear regression tends to have low variance, if the ratio -of \( n \) to \( p \) is moderately large. +
-Bootstrap aggregation, or just bagging, is a -general-purpose procedure for reducing the variance of a statistical -learning method. - -
-Bagging typically results in improved accuracy -over prediction using a single tree. Unfortunately, however, it can be -difficult to interpret the resulting model. Recall that one of the -advantages of decision trees is the attractive and easily interpreted -diagram that results. - -
-However, when we bag a large number of trees, it is no longer -possible to represent the resulting statistical learning procedure -using a single tree, and it is no longer clear which variables are -most important to the procedure. Thus, bagging improves prediction -accuracy at the expense of interpretability. Although the collection -of bagged trees is much more difficult to interpret than a single -tree, one can obtain an overall summary of the importance of each -predictor using the MSE (for bagging regression trees) or the Gini -index (for bagging classification trees). In the case of bagging -regression trees, we can record the total amount that the MSE is -decreased due to splits over a given predictor, averaged over all \( B \) possible -trees. A large value indicates an important predictor. Similarly, in -the context of bagging classification trees, we can add up the total -amount that the Gini index is decreased by splits over a given -predictor, averaged over all \( B \) trees. +However, by aggregating many decision trees, using methods like bagging, random forests, and boosting, the predictive performance of trees can be substantially improved.
@@ -234,6 +213,7 @@ predictor, averaged over all \( B \) trees.
+
+The plain decision trees suffer from high +variance. This means that if we split the training data into two parts +at random, and fit a decision tree to both halves, the results that we +get could be quite different. In contrast, a procedure with low +variance will yield similar results if applied repeatedly to distinct +data sets; linear regression tends to have low variance, if the ratio +of \( n \) to \( p \) is moderately large. + +
+Bootstrap aggregation, or just bagging, is a +general-purpose procedure for reducing the variance of a statistical +learning method. + +
+Bagging typically results in improved accuracy +over prediction using a single tree. Unfortunately, however, it can be +difficult to interpret the resulting model. Recall that one of the +advantages of decision trees is the attractive and easily interpreted +diagram that results. + +
+However, when we bag a large number of trees, it is no longer +possible to represent the resulting statistical learning procedure +using a single tree, and it is no longer clear which variables are +most important to the procedure. Thus, bagging improves prediction +accuracy at the expense of interpretability. Although the collection +of bagged trees is much more difficult to interpret than a single +tree, one can obtain an overall summary of the importance of each +predictor using the MSE (for bagging regression trees) or the Gini +index (for bagging classification trees). In the case of bagging +regression trees, we can record the total amount that the MSE is +decreased due to splits over a given predictor, averaged over all \( B \) possible +trees. A large value indicates an important predictor. Similarly, in +the context of bagging classification trees, we can add up the total +amount that the Gini index is decreased by splits over a given +predictor, averaged over all \( B \) trees. - -
heads_proba = 0.51
-coin_tosses = (np.random.rand(10000, 10) < heads_proba).astype(np.int32)
-cumulative_heads_ratio = np.cumsum(coin_tosses, axis=0) / np.arange(1, 10001).reshape(-1, 1)
-plt.figure(figsize=(8,3.5))
-plt.plot(cumulative_heads_ratio)
-plt.plot([0, 10000], [0.51, 0.51], "k--", linewidth=2, label="51%")
-plt.plot([0, 10000], [0.5, 0.5], "k-", label="50%")
-plt.xlabel("Number of coin tosses")
-plt.ylabel("Heads ratio")
-plt.legend(loc="lower right")
-plt.axis([0, 10000, 0.42, 0.58])
-plt.show()
-
@@ -210,6 +238,7 @@ plt.show()
-Random forests provide an improvement over bagged trees by way of a -small tweak that decorrelates the trees. - -
-As in bagging, we build a -number of decision trees on bootstrapped training samples. But when -building these decision trees, each time a split in a tree is -considered, a random sample of \( m \) predictors is chosen as split -candidates from the full set of \( p \) predictors. The split is allowed to -use only one of those \( m \) predictors. - -
-A fresh sample of \( m \) predictors is -taken at each split, and typically we choose - -$$ -m\approx \sqrt{p}. -$$ - -
-In building a random forest, at -each split in the tree, the algorithm is not even allowed to consider -a majority of the available predictors. - -
-The reason for this is rather clever. Suppose that there is one very -strong predictor in the data set, along with a number of other -moderately strong predictors. Then in the collection of bagged -variable importance random forest trees, most or all of the trees will -use this strong predictor in the top split. Consequently, all of the -bagged trees will look quite similar to each other. Hence the -predictions from the bagged trees will be highly correlated. -Unfortunately, averaging many highly correlated quantities does not -lead to as large of a reduction in variance as averaging many -uncorrelated quanti- ties. In particular, this means that bagging will -not lead to a substantial reduction in variance over a single tree in -this setting. + +
heads_proba = 0.51
+coin_tosses = (np.random.rand(10000, 10) < heads_proba).astype(np.int32)
+cumulative_heads_ratio = np.cumsum(coin_tosses, axis=0) / np.arange(1, 10001).reshape(-1, 1)
+plt.figure(figsize=(8,3.5))
+plt.plot(cumulative_heads_ratio)
+plt.plot([0, 10000], [0.51, 0.51], "k--", linewidth=2, label="51%")
+plt.plot([0, 10000], [0.5, 0.5], "k-", label="50%")
+plt.xlabel("Number of coin tosses")
+plt.ylabel("Heads ratio")
+plt.legend(loc="lower right")
+plt.axis([0, 10000, 0.42, 0.58])
+plt.show()
+
@@ -233,6 +214,7 @@ this setting.
+
+Random forests provide an improvement over bagged trees by way of a +small tweak that decorrelates the trees. + +
+As in bagging, we build a +number of decision trees on bootstrapped training samples. But when +building these decision trees, each time a split in a tree is +considered, a random sample of \( m \) predictors is chosen as split +candidates from the full set of \( p \) predictors. The split is allowed to +use only one of those \( m \) predictors. + +
+A fresh sample of \( m \) predictors is +taken at each split, and typically we choose + +$$ +m\approx \sqrt{p}. +$$ + +
+In building a random forest, at +each split in the tree, the algorithm is not even allowed to consider +a majority of the available predictors. + +
+The reason for this is rather clever. Suppose that there is one very +strong predictor in the data set, along with a number of other +moderately strong predictors. Then in the collection of bagged +variable importance random forest trees, most or all of the trees will +use this strong predictor in the top split. Consequently, all of the +bagged trees will look quite similar to each other. Hence the +predictions from the bagged trees will be highly correlated. +Unfortunately, averaging many highly correlated quantities does not +lead to as large of a reduction in variance as averaging many +uncorrelated quanti- ties. In particular, this means that bagging will +not lead to a substantial reduction in variance over a single tree in +this setting. - -
from sklearn.ensemble import RandomForestClassifier
-from sklearn.preprocessing import LabelEncoder
-from sklearn.model_selection import cross_validate
-# Data set not specificied
-X = dataset.XXX
-Y = dataset.YYY
-#Instantiate the model with 100 trees and entropy as splitting criteria
-Random_Forest_model = RandomForestClassifier(n_estimators=100,criterion="entropy")
-#Cross validation
-accuracy = cross_validate(Random_Forest_model,X,Y,cv=10)['test_score']
-
@@ -206,6 +237,7 @@ accuracy = cross_validate(Random_Forest_mode
-
from sklearn.model_selection import train_test_split
-from sklearn.datasets import make_moons
-
-X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
-X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
-from sklearn.ensemble import RandomForestClassifier
-from sklearn.ensemble import VotingClassifier
-from sklearn.linear_model import LogisticRegression
-from sklearn.svm import SVC
-
-log_clf = LogisticRegression(random_state=42)
-rnd_clf = RandomForestClassifier(random_state=42)
-svm_clf = SVC(random_state=42)
-
-voting_clf = VotingClassifier(
- estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
- voting='hard')
-voting_clf.fit(X_train, y_train)
-- - -
from sklearn.metrics import accuracy_score
-
-for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
- clf.fit(X_train, y_train)
- y_pred = clf.predict(X_test)
- print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
-- - -
log_clf = LogisticRegression(random_state=42)
-rnd_clf = RandomForestClassifier(random_state=42)
-svm_clf = SVC(probability=True, random_state=42)
-
-voting_clf = VotingClassifier(
- estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
- voting='soft')
-voting_clf.fit(X_train, y_train)
-- - -
from sklearn.metrics import accuracy_score
-
-for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
- clf.fit(X_train, y_train)
- y_pred = clf.predict(X_test)
- print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
+from sklearn.ensemble import RandomForestClassifier
+from sklearn.preprocessing import LabelEncoder
+from sklearn.model_selection import cross_validate
+# Data set not specificied
+X = dataset.XXX
+Y = dataset.YYY
+#Instantiate the model with 100 trees and entropy as splitting criteria
+Random_Forest_model = RandomForestClassifier(n_estimators=100,criterion="entropy")
+#Cross validation
+accuracy = cross_validate(Random_Forest_model,X,Y,cv=10)['test_score']
@@ -245,6 +210,7 @@ voting_clf.fit(X_train, y_train)
-
from sklearn.ensemble import BaggingClassifier
-from sklearn.tree import DecisionTreeClassifier
+from sklearn.model_selection import train_test_split
+from sklearn.datasets import make_moons
-bag_clf = BaggingClassifier(
- DecisionTreeClassifier(random_state=42), n_estimators=500,
- max_samples=100, bootstrap=True, n_jobs=-1, random_state=42)
-bag_clf.fit(X_train, y_train)
-y_pred = bag_clf.predict(X_test)
+X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
+X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
+from sklearn.ensemble import RandomForestClassifier
+from sklearn.ensemble import VotingClassifier
+from sklearn.linear_model import LogisticRegression
+from sklearn.svm import SVC
+
+log_clf = LogisticRegression(random_state=42)
+rnd_clf = RandomForestClassifier(random_state=42)
+svm_clf = SVC(random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='hard')
+voting_clf.fit(X_train, y_train)
from sklearn.metrics import accuracy_score
-print(accuracy_score(y_test, y_pred))
+
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
-
tree_clf = DecisionTreeClassifier(random_state=42)
-tree_clf.fit(X_train, y_train)
-y_pred_tree = tree_clf.predict(X_test)
-print(accuracy_score(y_test, y_pred_tree))
+log_clf = LogisticRegression(random_state=42)
+rnd_clf = RandomForestClassifier(random_state=42)
+svm_clf = SVC(probability=True, random_state=42)
+
+voting_clf = VotingClassifier(
+ estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
+ voting='soft')
+voting_clf.fit(X_train, y_train)
-
from matplotlib.colors import ListedColormap
+from sklearn.metrics import accuracy_score
-def plot_decision_boundary(clf, X, y, axes=[-1.5, 2.5, -1, 1.5], alpha=0.5, contour=True):
- x1s = np.linspace(axes[0], axes[1], 100)
- x2s = np.linspace(axes[2], axes[3], 100)
- x1, x2 = np.meshgrid(x1s, x2s)
- X_new = np.c_[x1.ravel(), x2.ravel()]
- y_pred = clf.predict(X_new).reshape(x1.shape)
- custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
- plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
- if contour:
- custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
- plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
- plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", alpha=alpha)
- plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", alpha=alpha)
- plt.axis(axes)
- plt.xlabel(r"$x_1$", fontsize=18)
- plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
-plt.figure(figsize=(11,4))
-plt.subplot(121)
-plot_decision_boundary(tree_clf, X, y)
-plt.title("Decision Tree", fontsize=14)
-plt.subplot(122)
-plot_decision_boundary(bag_clf, X, y)
-plt.title("Decision Trees with Bagging", fontsize=14)
-plt.show()
+for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
+ clf.fit(X_train, y_train)
+ y_pred = clf.predict(X_test)
+ print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
@@ -247,6 +249,7 @@ plt.show()
31
32
33
+ 34
»
diff --git a/doc/pub/DecisionTrees/html/._DecisionTrees-bs031.html b/doc/pub/DecisionTrees/html/._DecisionTrees-bs031.html
index 9b4825d3c..d85e0ae27 100644
--- a/doc/pub/DecisionTrees/html/._DecisionTrees-bs031.html
+++ b/doc/pub/DecisionTrees/html/._DecisionTrees-bs031.html
@@ -67,26 +67,30 @@ Automatically generated HTML file from DocOnce source
('A Classification Tree', 2, None, '___sec12'),
('Growing a classification tree', 2, None, '___sec13'),
('Classification tree, how to split nodes', 2, None, '___sec14'),
- ('Entropy and the ID3 algorithm', 2, None, '___sec15'),
- ('Implementing the ID3 Algorithm', 2, None, '___sec16'),
+ ('The CART (Classification and Regression Tree) algorithm',
+ 2,
+ None,
+ '___sec15'),
+ ('Entropy and the ID3 algorithm', 2, None, '___sec16'),
+ ('Implementing the ID3 Algorithm', 2, None, '___sec17'),
('Cancer Data again now with Decision Trees',
2,
None,
- '___sec17'),
- ('Another example, the moons again', 2, None, '___sec18'),
- ('Playing around with regions', 2, None, '___sec19'),
- ('Regression trees', 2, None, '___sec20'),
- ('Final regressor code', 2, None, '___sec21'),
- ('Pros and cons of trees, pros', 2, None, '___sec22'),
- ('Disadvantages', 2, None, '___sec23'),
- ('Bagging', 2, None, '___sec24'),
- ('Simple example, head or tail', 2, None, '___sec25'),
- ('Random forests', 2, None, '___sec26'),
- ('A simple scikit-learn example', 2, None, '___sec27'),
- ('Please, not the moons again!', 2, None, '___sec28'),
- ('Bagging examples', 2, None, '___sec29'),
- ('Then random forests', 2, None, '___sec30'),
- ('Boosting and more', 2, None, '___sec31')]}
+ '___sec18'),
+ ('Another example, the moons again', 2, None, '___sec19'),
+ ('Playing around with regions', 2, None, '___sec20'),
+ ('Regression trees', 2, None, '___sec21'),
+ ('Final regressor code', 2, None, '___sec22'),
+ ('Pros and cons of trees, pros', 2, None, '___sec23'),
+ ('Disadvantages', 2, None, '___sec24'),
+ ('Bagging', 2, None, '___sec25'),
+ ('Simple example, head or tail', 2, None, '___sec26'),
+ ('Random forests', 2, None, '___sec27'),
+ ('A simple scikit-learn example', 2, None, '___sec28'),
+ ('Please, not the moons again!', 2, None, '___sec29'),
+ ('Bagging examples', 2, None, '___sec30'),
+ ('Then random forests', 2, None, '___sec31'),
+ ('Boosting and more', 2, None, '___sec32')]}
end of tocinfo -->
@@ -139,23 +143,24 @@ MathJax.Hub.Config({
A Classification Tree
Growing a classification tree
Classification tree, how to split nodes
- Entropy and the ID3 algorithm
- Implementing the ID3 Algorithm
- Cancer Data again now with Decision Trees
- Another example, the moons again
- Playing around with regions
- Regression trees
- Final regressor code
- Pros and cons of trees, pros
- Disadvantages
- Bagging
- Simple example, head or tail
- Random forests
- A simple scikit-learn example
- Please, not the moons again!
- Bagging examples
- Then random forests
- Boosting and more
+ The CART (Classification and Regression Tree) algorithm
+ Entropy and the ID3 algorithm
+ Implementing the ID3 Algorithm
+ Cancer Data again now with Decision Trees
+ Another example, the moons again
+ Playing around with regions
+ Regression trees
+ Final regressor code
+ Pros and cons of trees, pros
+ Disadvantages
+ Bagging
+ Simple example, head or tail
+ Random forests
+ A simple scikit-learn example
+ Please, not the moons again!
+ Bagging examples
+ Then random forests
+ Boosting and more
@@ -171,24 +176,63 @@ MathJax.Hub.Config({
-Then random forests
+Bagging examples
+
-
bag_clf = BaggingClassifier(
- DecisionTreeClassifier(splitter="random", max_leaf_nodes=16, random_state=42),
- n_estimators=500, max_samples=1.0, bootstrap=True, n_jobs=-1, random_state=42)
+from sklearn.ensemble import BaggingClassifier
+from sklearn.tree import DecisionTreeClassifier
+
+bag_clf = BaggingClassifier(
+ DecisionTreeClassifier(random_state=42), n_estimators=500,
+ max_samples=100, bootstrap=True, n_jobs=-1, random_state=42)
+bag_clf.fit(X_train, y_train)
+y_pred = bag_clf.predict(X_test)
-
bag_clf.fit(X_train, y_train)
-y_pred = bag_clf.predict(X_test)
-from sklearn.ensemble import RandomForestClassifier
-rnd_clf = RandomForestClassifier(n_estimators=500, max_leaf_nodes=16, n_jobs=-1, random_state=42)
-rnd_clf.fit(X_train, y_train)
-y_pred_rf = rnd_clf.predict(X_test)
-np.sum(y_pred == y_pred_rf) / len(y_pred)
+from sklearn.metrics import accuracy_score
+print(accuracy_score(y_test, y_pred))
+
+
+
+
+
tree_clf = DecisionTreeClassifier(random_state=42)
+tree_clf.fit(X_train, y_train)
+y_pred_tree = tree_clf.predict(X_test)
+print(accuracy_score(y_test, y_pred_tree))
+
+
+
+
+
from matplotlib.colors import ListedColormap
+
+def plot_decision_boundary(clf, X, y, axes=[-1.5, 2.5, -1, 1.5], alpha=0.5, contour=True):
+ x1s = np.linspace(axes[0], axes[1], 100)
+ x2s = np.linspace(axes[2], axes[3], 100)
+ x1, x2 = np.meshgrid(x1s, x2s)
+ X_new = np.c_[x1.ravel(), x2.ravel()]
+ y_pred = clf.predict(X_new).reshape(x1.shape)
+ custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
+ plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
+ if contour:
+ custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
+ plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
+ plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", alpha=alpha)
+ plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", alpha=alpha)
+ plt.axis(axes)
+ plt.xlabel(r"$x_1$", fontsize=18)
+ plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
+plt.figure(figsize=(11,4))
+plt.subplot(121)
+plot_decision_boundary(tree_clf, X, y)
+plt.title("Decision Tree", fontsize=14)
+plt.subplot(122)
+plot_decision_boundary(bag_clf, X, y)
+plt.title("Decision Trees with Bagging", fontsize=14)
+plt.show()
@@ -207,6 +251,7 @@ np.sum(y_pred =
31
32
33
+ 34
»
diff --git a/doc/pub/DecisionTrees/html/DecisionTrees-bs.html b/doc/pub/DecisionTrees/html/DecisionTrees-bs.html
index ceb0e0248..8136086bf 100644
--- a/doc/pub/DecisionTrees/html/DecisionTrees-bs.html
+++ b/doc/pub/DecisionTrees/html/DecisionTrees-bs.html
@@ -67,26 +67,30 @@ Automatically generated HTML file from DocOnce source
('A Classification Tree', 2, None, '___sec12'),
('Growing a classification tree', 2, None, '___sec13'),
('Classification tree, how to split nodes', 2, None, '___sec14'),
- ('Entropy and the ID3 algorithm', 2, None, '___sec15'),
- ('Implementing the ID3 Algorithm', 2, None, '___sec16'),
+ ('The CART (Classification and Regression Tree) algorithm',
+ 2,
+ None,
+ '___sec15'),
+ ('Entropy and the ID3 algorithm', 2, None, '___sec16'),
+ ('Implementing the ID3 Algorithm', 2, None, '___sec17'),
('Cancer Data again now with Decision Trees',
2,
None,
- '___sec17'),
- ('Another example, the moons again', 2, None, '___sec18'),
- ('Playing around with regions', 2, None, '___sec19'),
- ('Regression trees', 2, None, '___sec20'),
- ('Final regressor code', 2, None, '___sec21'),
- ('Pros and cons of trees, pros', 2, None, '___sec22'),
- ('Disadvantages', 2, None, '___sec23'),
- ('Bagging', 2, None, '___sec24'),
- ('Simple example, head or tail', 2, None, '___sec25'),
- ('Random forests', 2, None, '___sec26'),
- ('A simple scikit-learn example', 2, None, '___sec27'),
- ('Please, not the moons again!', 2, None, '___sec28'),
- ('Bagging examples', 2, None, '___sec29'),
- ('Then random forests', 2, None, '___sec30'),
- ('Boosting and more', 2, None, '___sec31')]}
+ '___sec18'),
+ ('Another example, the moons again', 2, None, '___sec19'),
+ ('Playing around with regions', 2, None, '___sec20'),
+ ('Regression trees', 2, None, '___sec21'),
+ ('Final regressor code', 2, None, '___sec22'),
+ ('Pros and cons of trees, pros', 2, None, '___sec23'),
+ ('Disadvantages', 2, None, '___sec24'),
+ ('Bagging', 2, None, '___sec25'),
+ ('Simple example, head or tail', 2, None, '___sec26'),
+ ('Random forests', 2, None, '___sec27'),
+ ('A simple scikit-learn example', 2, None, '___sec28'),
+ ('Please, not the moons again!', 2, None, '___sec29'),
+ ('Bagging examples', 2, None, '___sec30'),
+ ('Then random forests', 2, None, '___sec31'),
+ ('Boosting and more', 2, None, '___sec32')]}
end of tocinfo -->
@@ -139,23 +143,24 @@ MathJax.Hub.Config({
A Classification Tree
Growing a classification tree
Classification tree, how to split nodes
- Entropy and the ID3 algorithm
- Implementing the ID3 Algorithm
- Cancer Data again now with Decision Trees
- Another example, the moons again
- Playing around with regions
- Regression trees
- Final regressor code
- Pros and cons of trees, pros
- Disadvantages
- Bagging
- Simple example, head or tail
- Random forests
- A simple scikit-learn example
- Please, not the moons again!
- Bagging examples
- Then random forests
- Boosting and more
+ The CART (Classification and Regression Tree) algorithm
+ Entropy and the ID3 algorithm
+ Implementing the ID3 Algorithm
+ Cancer Data again now with Decision Trees
+ Another example, the moons again
+ Playing around with regions
+ Regression trees
+ Final regressor code
+ Pros and cons of trees, pros
+ Disadvantages
+ Bagging
+ Simple example, head or tail
+ Random forests
+ A simple scikit-learn example
+ Please, not the moons again!
+ Bagging examples
+ Then random forests
+ Boosting and more
@@ -190,7 +195,7 @@ MathJax.Hub.Config({
[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University
-
Oct 26, 2019
+Oct 29, 2019
@@ -214,7 +219,7 @@ MathJax.Hub.Config({
9
10
...
- 33
+ 34
»
diff --git a/doc/pub/DecisionTrees/html/DecisionTrees-reveal.html b/doc/pub/DecisionTrees/html/DecisionTrees-reveal.html
index cf3dc3ebb..aca451881 100644
--- a/doc/pub/DecisionTrees/html/DecisionTrees-reveal.html
+++ b/doc/pub/DecisionTrees/html/DecisionTrees-reveal.html
@@ -148,7 +148,7 @@ MathJax.Hub.Config({
[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University
-
Oct 26, 2019
+Oct 29, 2019
@@ -627,7 +627,191 @@ $$
-Entropy and the ID3 algorithm
+The CART (Classification and Regression Tree) algorithm
+
+
+The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3.
+
+
+
+
+
from random import seed
+from random import randrange
+from csv import reader
+
+# Load a CSV file
+def load_csv(filename):
+ file = open(filename, "rb")
+ lines = reader(file)
+ dataset = list(lines)
+ return dataset
+
+# Convert string column to float
+def str_column_to_float(dataset, column):
+ for row in dataset:
+ row[column] = float(row[column].strip())
+
+# Split a dataset into k folds
+def cross_validation_split(dataset, n_folds):
+ dataset_split = list()
+ dataset_copy = list(dataset)
+ fold_size = int(len(dataset) / n_folds)
+ for i in range(n_folds):
+ fold = list()
+ while len(fold) < fold_size:
+ index = randrange(len(dataset_copy))
+ fold.append(dataset_copy.pop(index))
+ dataset_split.append(fold)
+ return dataset_split
+
+# Calculate accuracy percentage
+def accuracy_metric(actual, predicted):
+ correct = 0
+ for i in range(len(actual)):
+ if actual[i] == predicted[i]:
+ correct += 1
+ return correct / float(len(actual)) * 100.0
+
+# Evaluate an algorithm using a cross validation split
+def evaluate_algorithm(dataset, algorithm, n_folds, *args):
+ folds = cross_validation_split(dataset, n_folds)
+ scores = list()
+ for fold in folds:
+ train_set = list(folds)
+ train_set.remove(fold)
+ train_set = sum(train_set, [])
+ test_set = list()
+ for row in fold:
+ row_copy = list(row)
+ test_set.append(row_copy)
+ row_copy[-1] = None
+ predicted = algorithm(train_set, test_set, *args)
+ actual = [row[-1] for row in fold]
+ accuracy = accuracy_metric(actual, predicted)
+ scores.append(accuracy)
+ return scores
+
+# Split a dataset based on an attribute and an attribute value
+def test_split(index, value, dataset):
+ left, right = list(), list()
+ for row in dataset:
+ if row[index] < value:
+ left.append(row)
+ else:
+ right.append(row)
+ return left, right
+
+# Calculate the Gini index for a split dataset
+def gini_index(groups, classes):
+ # count all samples at split point
+ n_instances = float(sum([len(group) for group in groups]))
+ # sum weighted Gini index for each group
+ gini = 0.0
+ for group in groups:
+ size = float(len(group))
+ # avoid divide by zero
+ if size == 0:
+ continue
+ score = 0.0
+ # score the group based on the score for each class
+ for class_val in classes:
+ p = [row[-1] for row in group].count(class_val) / size
+ score += p * p
+ # weight the group score by its relative size
+ gini += (1.0 - score) * (size / n_instances)
+ return gini
+
+# Select the best split point for a dataset
+def get_split(dataset):
+ class_values = list(set(row[-1] for row in dataset))
+ b_index, b_value, b_score, b_groups = 999, 999, 999, None
+ for index in range(len(dataset[0])-1):
+ for row in dataset:
+ groups = test_split(index, row[index], dataset)
+ gini = gini_index(groups, class_values)
+ if gini < b_score:
+ b_index, b_value, b_score, b_groups = index, row[index], gini, groups
+ return {'index':b_index, 'value':b_value, 'groups':b_groups}
+
+# Create a terminal node value
+def to_terminal(group):
+ outcomes = [row[-1] for row in group]
+ return max(set(outcomes), key=outcomes.count)
+
+# Create child splits for a node or make terminal
+def split(node, max_depth, min_size, depth):
+ left, right = node['groups']
+ del(node['groups'])
+ # check for a no split
+ if not left or not right:
+ node['left'] = node['right'] = to_terminal(left + right)
+ return
+ # check for max depth
+ if depth >= max_depth:
+ node['left'], node['right'] = to_terminal(left), to_terminal(right)
+ return
+ # process left child
+ if len(left) <= min_size:
+ node['left'] = to_terminal(left)
+ else:
+ node['left'] = get_split(left)
+ split(node['left'], max_depth, min_size, depth+1)
+ # process right child
+ if len(right) <= min_size:
+ node['right'] = to_terminal(right)
+ else:
+ node['right'] = get_split(right)
+ split(node['right'], max_depth, min_size, depth+1)
+
+# Build a decision tree
+def build_tree(train, max_depth, min_size):
+ root = get_split(train)
+ split(root, max_depth, min_size, 1)
+ return root
+
+# Make a prediction with a decision tree
+def predict(node, row):
+ if row[node['index']] < node['value']:
+ if isinstance(node['left'], dict):
+ return predict(node['left'], row)
+ else:
+ return node['left']
+ else:
+ if isinstance(node['right'], dict):
+ return predict(node['right'], row)
+ else:
+ return node['right']
+
+# Classification and Regression Tree Algorithm
+def decision_tree(train, test, max_depth, min_size):
+ tree = build_tree(train, max_depth, min_size)
+ predictions = list()
+ for row in test:
+ prediction = predict(tree, row)
+ predictions.append(prediction)
+ return(predictions)
+
+# Test CART
+seed(1)
+# load and prepare data
+filename = 'DataFiles/rideclass.csv'
+dataset = load_csv(filename)
+# convert string attributes to integers
+for i in range(len(dataset[0])):
+ str_column_to_float(dataset, i)
+# evaluate algorithm
+n_folds = 5
+max_depth = 5
+min_size = 10
+scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
+print('Scores: %s' % scores)
+print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+
+
+
+
+
+Entropy and the ID3 algorithm
ID3, learns decision trees by constructing
@@ -664,7 +848,7 @@ attributes at each step while growing the tree.
-Implementing the ID3 Algorithm
+Implementing the ID3 Algorithm
more text to come here, material presented during lecture Friday Oct 25.
@@ -672,7 +856,7 @@ attributes at each step while growing the tree.
-Cancer Data again now with Decision Trees
+Cancer Data again now with Decision Trees
@@ -722,7 +906,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
-Another example, the moons again
+Another example, the moons again
@@ -795,7 +979,7 @@ plt.show()
-Playing around with regions
+Playing around with regions
@@ -824,7 +1008,7 @@ plt.show()
-Regression trees
+Regression trees
@@ -847,7 +1031,7 @@ tree_reg.fit(X, y)
-Final regressor code
+Final regressor code
@@ -926,7 +1110,7 @@ plt.show()
-Pros and cons of trees, pros
+Pros and cons of trees, pros
- White box, easy to interpret model. Some people believe that decision trees more closely mirror human decision-making than do the regression and classification approaches discussed earlier (think of support vector machines)
@@ -941,7 +1125,7 @@ plt.show()
-Disadvantages
+Disadvantages
- Unfortunately, trees generally do not have the same level of predictive accuracy as some of the other regression and classification approaches
@@ -959,7 +1143,7 @@ However, by aggregating many decision trees, using methods like bagging, random
-Bagging
+Bagging
The plain decision trees suffer from high
@@ -1002,7 +1186,7 @@ predictor, averaged over all \( B \) trees.
-Simple example, head or tail
+Simple example, head or tail
@@ -1023,7 +1207,7 @@ plt.show()
-Random forests
+Random forests
Random forests provide an improvement over bagged trees by way of a
@@ -1069,7 +1253,7 @@ this setting.
-A simple scikit-learn example
+A simple scikit-learn example
@@ -1088,7 +1272,7 @@ accuracy = cross_validate(Random_Forest_model,X,Y,cv=Please, not the moons again!
+Please, not the moons again!
@@ -1147,7 +1331,7 @@ voting_clf.fit(X_train, y_train)
-Bagging examples
+Bagging examples
@@ -1209,7 +1393,7 @@ plt.show()
-Then random forests
+Then random forests
@@ -1232,7 +1416,7 @@ np.sum(y_pred == y_pred_rf) / len(y_pred)
-Boosting and more
+Boosting and more
More material to come here.
diff --git a/doc/pub/DecisionTrees/html/DecisionTrees-solarized.html b/doc/pub/DecisionTrees/html/DecisionTrees-solarized.html
index f6b84cb41..c889b56e0 100644
--- a/doc/pub/DecisionTrees/html/DecisionTrees-solarized.html
+++ b/doc/pub/DecisionTrees/html/DecisionTrees-solarized.html
@@ -87,26 +87,30 @@ div { text-align: justify; text-justify: inter-word; }
('A Classification Tree', 2, None, '___sec12'),
('Growing a classification tree', 2, None, '___sec13'),
('Classification tree, how to split nodes', 2, None, '___sec14'),
- ('Entropy and the ID3 algorithm', 2, None, '___sec15'),
- ('Implementing the ID3 Algorithm', 2, None, '___sec16'),
+ ('The CART (Classification and Regression Tree) algorithm',
+ 2,
+ None,
+ '___sec15'),
+ ('Entropy and the ID3 algorithm', 2, None, '___sec16'),
+ ('Implementing the ID3 Algorithm', 2, None, '___sec17'),
('Cancer Data again now with Decision Trees',
2,
None,
- '___sec17'),
- ('Another example, the moons again', 2, None, '___sec18'),
- ('Playing around with regions', 2, None, '___sec19'),
- ('Regression trees', 2, None, '___sec20'),
- ('Final regressor code', 2, None, '___sec21'),
- ('Pros and cons of trees, pros', 2, None, '___sec22'),
- ('Disadvantages', 2, None, '___sec23'),
- ('Bagging', 2, None, '___sec24'),
- ('Simple example, head or tail', 2, None, '___sec25'),
- ('Random forests', 2, None, '___sec26'),
- ('A simple scikit-learn example', 2, None, '___sec27'),
- ('Please, not the moons again!', 2, None, '___sec28'),
- ('Bagging examples', 2, None, '___sec29'),
- ('Then random forests', 2, None, '___sec30'),
- ('Boosting and more', 2, None, '___sec31')]}
+ '___sec18'),
+ ('Another example, the moons again', 2, None, '___sec19'),
+ ('Playing around with regions', 2, None, '___sec20'),
+ ('Regression trees', 2, None, '___sec21'),
+ ('Final regressor code', 2, None, '___sec22'),
+ ('Pros and cons of trees, pros', 2, None, '___sec23'),
+ ('Disadvantages', 2, None, '___sec24'),
+ ('Bagging', 2, None, '___sec25'),
+ ('Simple example, head or tail', 2, None, '___sec26'),
+ ('Random forests', 2, None, '___sec27'),
+ ('A simple scikit-learn example', 2, None, '___sec28'),
+ ('Please, not the moons again!', 2, None, '___sec29'),
+ ('Bagging examples', 2, None, '___sec30'),
+ ('Then random forests', 2, None, '___sec31'),
+ ('Boosting and more', 2, None, '___sec32')]}
end of tocinfo -->
@@ -148,7 +152,7 @@ MathJax.Hub.Config({
[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University
-
Oct 26, 2019
+Oct 29, 2019
@@ -602,7 +606,190 @@ $$
-
Entropy and the ID3 algorithm
+The CART (Classification and Regression Tree) algorithm
+
+
+The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3.
+
+
+
+
+
from random import seed
+from random import randrange
+from csv import reader
+
+# Load a CSV file
+def load_csv(filename):
+ file = open(filename, "rb")
+ lines = reader(file)
+ dataset = list(lines)
+ return dataset
+
+# Convert string column to float
+def str_column_to_float(dataset, column):
+ for row in dataset:
+ row[column] = float(row[column].strip())
+
+# Split a dataset into k folds
+def cross_validation_split(dataset, n_folds):
+ dataset_split = list()
+ dataset_copy = list(dataset)
+ fold_size = int(len(dataset) / n_folds)
+ for i in range(n_folds):
+ fold = list()
+ while len(fold) < fold_size:
+ index = randrange(len(dataset_copy))
+ fold.append(dataset_copy.pop(index))
+ dataset_split.append(fold)
+ return dataset_split
+
+# Calculate accuracy percentage
+def accuracy_metric(actual, predicted):
+ correct = 0
+ for i in range(len(actual)):
+ if actual[i] == predicted[i]:
+ correct += 1
+ return correct / float(len(actual)) * 100.0
+
+# Evaluate an algorithm using a cross validation split
+def evaluate_algorithm(dataset, algorithm, n_folds, *args):
+ folds = cross_validation_split(dataset, n_folds)
+ scores = list()
+ for fold in folds:
+ train_set = list(folds)
+ train_set.remove(fold)
+ train_set = sum(train_set, [])
+ test_set = list()
+ for row in fold:
+ row_copy = list(row)
+ test_set.append(row_copy)
+ row_copy[-1] = None
+ predicted = algorithm(train_set, test_set, *args)
+ actual = [row[-1] for row in fold]
+ accuracy = accuracy_metric(actual, predicted)
+ scores.append(accuracy)
+ return scores
+
+# Split a dataset based on an attribute and an attribute value
+def test_split(index, value, dataset):
+ left, right = list(), list()
+ for row in dataset:
+ if row[index] < value:
+ left.append(row)
+ else:
+ right.append(row)
+ return left, right
+
+# Calculate the Gini index for a split dataset
+def gini_index(groups, classes):
+ # count all samples at split point
+ n_instances = float(sum([len(group) for group in groups]))
+ # sum weighted Gini index for each group
+ gini = 0.0
+ for group in groups:
+ size = float(len(group))
+ # avoid divide by zero
+ if size == 0:
+ continue
+ score = 0.0
+ # score the group based on the score for each class
+ for class_val in classes:
+ p = [row[-1] for row in group].count(class_val) / size
+ score += p * p
+ # weight the group score by its relative size
+ gini += (1.0 - score) * (size / n_instances)
+ return gini
+
+# Select the best split point for a dataset
+def get_split(dataset):
+ class_values = list(set(row[-1] for row in dataset))
+ b_index, b_value, b_score, b_groups = 999, 999, 999, None
+ for index in range(len(dataset[0])-1):
+ for row in dataset:
+ groups = test_split(index, row[index], dataset)
+ gini = gini_index(groups, class_values)
+ if gini < b_score:
+ b_index, b_value, b_score, b_groups = index, row[index], gini, groups
+ return {'index':b_index, 'value':b_value, 'groups':b_groups}
+
+# Create a terminal node value
+def to_terminal(group):
+ outcomes = [row[-1] for row in group]
+ return max(set(outcomes), key=outcomes.count)
+
+# Create child splits for a node or make terminal
+def split(node, max_depth, min_size, depth):
+ left, right = node['groups']
+ del(node['groups'])
+ # check for a no split
+ if not left or not right:
+ node['left'] = node['right'] = to_terminal(left + right)
+ return
+ # check for max depth
+ if depth >= max_depth:
+ node['left'], node['right'] = to_terminal(left), to_terminal(right)
+ return
+ # process left child
+ if len(left) <= min_size:
+ node['left'] = to_terminal(left)
+ else:
+ node['left'] = get_split(left)
+ split(node['left'], max_depth, min_size, depth+1)
+ # process right child
+ if len(right) <= min_size:
+ node['right'] = to_terminal(right)
+ else:
+ node['right'] = get_split(right)
+ split(node['right'], max_depth, min_size, depth+1)
+
+# Build a decision tree
+def build_tree(train, max_depth, min_size):
+ root = get_split(train)
+ split(root, max_depth, min_size, 1)
+ return root
+
+# Make a prediction with a decision tree
+def predict(node, row):
+ if row[node['index']] < node['value']:
+ if isinstance(node['left'], dict):
+ return predict(node['left'], row)
+ else:
+ return node['left']
+ else:
+ if isinstance(node['right'], dict):
+ return predict(node['right'], row)
+ else:
+ return node['right']
+
+# Classification and Regression Tree Algorithm
+def decision_tree(train, test, max_depth, min_size):
+ tree = build_tree(train, max_depth, min_size)
+ predictions = list()
+ for row in test:
+ prediction = predict(tree, row)
+ predictions.append(prediction)
+ return(predictions)
+
+# Test CART
+seed(1)
+# load and prepare data
+filename = 'DataFiles/rideclass.csv'
+dataset = load_csv(filename)
+# convert string attributes to integers
+for i in range(len(dataset[0])):
+ str_column_to_float(dataset, i)
+# evaluate algorithm
+n_folds = 5
+max_depth = 5
+min_size = 10
+scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
+print('Scores: %s' % scores)
+print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+
+
+
+
+
Entropy and the ID3 algorithm
ID3, learns decision trees by constructing
@@ -638,7 +825,7 @@ attributes at each step while growing the tree.
-
Implementing the ID3 Algorithm
+Implementing the ID3 Algorithm
more text to come here, material presented during lecture Friday Oct 25.
@@ -646,7 +833,7 @@ attributes at each step while growing the tree.
-
Cancer Data again now with Decision Trees
+Cancer Data again now with Decision Trees
@@ -695,7 +882,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
-
Another example, the moons again
+Another example, the moons again
@@ -767,7 +954,7 @@ plt.show()
-
Playing around with regions
+Playing around with regions
@@ -795,7 +982,7 @@ plt.show()
-
Regression trees
+Regression trees
@@ -817,7 +1004,7 @@ tree_reg.fit(X, y)
-
Final regressor code
+Final regressor code
@@ -895,7 +1082,7 @@ plt.show()
-
Pros and cons of trees, pros
+Pros and cons of trees, pros
- White box, easy to interpret model. Some people believe that decision trees more closely mirror human decision-making than do the regression and classification approaches discussed earlier (think of support vector machines)
@@ -909,7 +1096,7 @@ plt.show()
-Disadvantages
+Disadvantages
- Unfortunately, trees generally do not have the same level of predictive accuracy as some of the other regression and classification approaches
@@ -926,7 +1113,7 @@ However, by aggregating many decision trees, using methods like bagging, random
-
Bagging
+Bagging
The plain decision trees suffer from high
@@ -969,7 +1156,7 @@ predictor, averaged over all \( B \) trees.
-
Simple example, head or tail
+Simple example, head or tail
@@ -989,7 +1176,7 @@ plt.show()
-
Random forests
+Random forests
Random forests provide an improvement over bagged trees by way of a
@@ -1033,7 +1220,7 @@ this setting.
-
A simple scikit-learn example
+A simple scikit-learn example
@@ -1051,7 +1238,7 @@ accuracy = cross_validate(Random_Forest_model,X,Y,cv=Please, not the moons again!
+Please, not the moons again!
@@ -1109,7 +1296,7 @@ voting_clf.fit(X_train, y_train)
-
Bagging examples
+Bagging examples
@@ -1170,7 +1357,7 @@ plt.show()
-
Then random forests
+Then random forests
@@ -1192,7 +1379,7 @@ np.sum(y_pred == y_pred_rf) / len(y_pred)
-
Boosting and more
+Boosting and more
More material to come here.
diff --git a/doc/pub/DecisionTrees/html/DecisionTrees.html b/doc/pub/DecisionTrees/html/DecisionTrees.html
index 1472b9062..dfb71089a 100644
--- a/doc/pub/DecisionTrees/html/DecisionTrees.html
+++ b/doc/pub/DecisionTrees/html/DecisionTrees.html
@@ -92,26 +92,30 @@ div { text-align: justify; text-justify: inter-word; }
('A Classification Tree', 2, None, '___sec12'),
('Growing a classification tree', 2, None, '___sec13'),
('Classification tree, how to split nodes', 2, None, '___sec14'),
- ('Entropy and the ID3 algorithm', 2, None, '___sec15'),
- ('Implementing the ID3 Algorithm', 2, None, '___sec16'),
+ ('The CART (Classification and Regression Tree) algorithm',
+ 2,
+ None,
+ '___sec15'),
+ ('Entropy and the ID3 algorithm', 2, None, '___sec16'),
+ ('Implementing the ID3 Algorithm', 2, None, '___sec17'),
('Cancer Data again now with Decision Trees',
2,
None,
- '___sec17'),
- ('Another example, the moons again', 2, None, '___sec18'),
- ('Playing around with regions', 2, None, '___sec19'),
- ('Regression trees', 2, None, '___sec20'),
- ('Final regressor code', 2, None, '___sec21'),
- ('Pros and cons of trees, pros', 2, None, '___sec22'),
- ('Disadvantages', 2, None, '___sec23'),
- ('Bagging', 2, None, '___sec24'),
- ('Simple example, head or tail', 2, None, '___sec25'),
- ('Random forests', 2, None, '___sec26'),
- ('A simple scikit-learn example', 2, None, '___sec27'),
- ('Please, not the moons again!', 2, None, '___sec28'),
- ('Bagging examples', 2, None, '___sec29'),
- ('Then random forests', 2, None, '___sec30'),
- ('Boosting and more', 2, None, '___sec31')]}
+ '___sec18'),
+ ('Another example, the moons again', 2, None, '___sec19'),
+ ('Playing around with regions', 2, None, '___sec20'),
+ ('Regression trees', 2, None, '___sec21'),
+ ('Final regressor code', 2, None, '___sec22'),
+ ('Pros and cons of trees, pros', 2, None, '___sec23'),
+ ('Disadvantages', 2, None, '___sec24'),
+ ('Bagging', 2, None, '___sec25'),
+ ('Simple example, head or tail', 2, None, '___sec26'),
+ ('Random forests', 2, None, '___sec27'),
+ ('A simple scikit-learn example', 2, None, '___sec28'),
+ ('Please, not the moons again!', 2, None, '___sec29'),
+ ('Bagging examples', 2, None, '___sec30'),
+ ('Then random forests', 2, None, '___sec31'),
+ ('Boosting and more', 2, None, '___sec32')]}
end of tocinfo -->
@@ -153,7 +157,7 @@ MathJax.Hub.Config({
[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University
-
Oct 26, 2019
+Oct 29, 2019
@@ -607,7 +611,190 @@ $$
-
Entropy and the ID3 algorithm
+The CART (Classification and Regression Tree) algorithm
+
+
+The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3.
+
+
+
+
+
from random import seed
+from random import randrange
+from csv import reader
+
+# Load a CSV file
+def load_csv(filename):
+ file = open(filename, "rb")
+ lines = reader(file)
+ dataset = list(lines)
+ return dataset
+
+# Convert string column to float
+def str_column_to_float(dataset, column):
+ for row in dataset:
+ row[column] = float(row[column].strip())
+
+# Split a dataset into k folds
+def cross_validation_split(dataset, n_folds):
+ dataset_split = list()
+ dataset_copy = list(dataset)
+ fold_size = int(len(dataset) / n_folds)
+ for i in range(n_folds):
+ fold = list()
+ while len(fold) < fold_size:
+ index = randrange(len(dataset_copy))
+ fold.append(dataset_copy.pop(index))
+ dataset_split.append(fold)
+ return dataset_split
+
+# Calculate accuracy percentage
+def accuracy_metric(actual, predicted):
+ correct = 0
+ for i in range(len(actual)):
+ if actual[i] == predicted[i]:
+ correct += 1
+ return correct / float(len(actual)) * 100.0
+
+# Evaluate an algorithm using a cross validation split
+def evaluate_algorithm(dataset, algorithm, n_folds, *args):
+ folds = cross_validation_split(dataset, n_folds)
+ scores = list()
+ for fold in folds:
+ train_set = list(folds)
+ train_set.remove(fold)
+ train_set = sum(train_set, [])
+ test_set = list()
+ for row in fold:
+ row_copy = list(row)
+ test_set.append(row_copy)
+ row_copy[-1] = None
+ predicted = algorithm(train_set, test_set, *args)
+ actual = [row[-1] for row in fold]
+ accuracy = accuracy_metric(actual, predicted)
+ scores.append(accuracy)
+ return scores
+
+# Split a dataset based on an attribute and an attribute value
+def test_split(index, value, dataset):
+ left, right = list(), list()
+ for row in dataset:
+ if row[index] < value:
+ left.append(row)
+ else:
+ right.append(row)
+ return left, right
+
+# Calculate the Gini index for a split dataset
+def gini_index(groups, classes):
+ # count all samples at split point
+ n_instances = float(sum([len(group) for group in groups]))
+ # sum weighted Gini index for each group
+ gini = 0.0
+ for group in groups:
+ size = float(len(group))
+ # avoid divide by zero
+ if size == 0:
+ continue
+ score = 0.0
+ # score the group based on the score for each class
+ for class_val in classes:
+ p = [row[-1] for row in group].count(class_val) / size
+ score += p * p
+ # weight the group score by its relative size
+ gini += (1.0 - score) * (size / n_instances)
+ return gini
+
+# Select the best split point for a dataset
+def get_split(dataset):
+ class_values = list(set(row[-1] for row in dataset))
+ b_index, b_value, b_score, b_groups = 999, 999, 999, None
+ for index in range(len(dataset[0])-1):
+ for row in dataset:
+ groups = test_split(index, row[index], dataset)
+ gini = gini_index(groups, class_values)
+ if gini < b_score:
+ b_index, b_value, b_score, b_groups = index, row[index], gini, groups
+ return {'index':b_index, 'value':b_value, 'groups':b_groups}
+
+# Create a terminal node value
+def to_terminal(group):
+ outcomes = [row[-1] for row in group]
+ return max(set(outcomes), key=outcomes.count)
+
+# Create child splits for a node or make terminal
+def split(node, max_depth, min_size, depth):
+ left, right = node['groups']
+ del(node['groups'])
+ # check for a no split
+ if not left or not right:
+ node['left'] = node['right'] = to_terminal(left + right)
+ return
+ # check for max depth
+ if depth >= max_depth:
+ node['left'], node['right'] = to_terminal(left), to_terminal(right)
+ return
+ # process left child
+ if len(left) <= min_size:
+ node['left'] = to_terminal(left)
+ else:
+ node['left'] = get_split(left)
+ split(node['left'], max_depth, min_size, depth+1)
+ # process right child
+ if len(right) <= min_size:
+ node['right'] = to_terminal(right)
+ else:
+ node['right'] = get_split(right)
+ split(node['right'], max_depth, min_size, depth+1)
+
+# Build a decision tree
+def build_tree(train, max_depth, min_size):
+ root = get_split(train)
+ split(root, max_depth, min_size, 1)
+ return root
+
+# Make a prediction with a decision tree
+def predict(node, row):
+ if row[node['index']] < node['value']:
+ if isinstance(node['left'], dict):
+ return predict(node['left'], row)
+ else:
+ return node['left']
+ else:
+ if isinstance(node['right'], dict):
+ return predict(node['right'], row)
+ else:
+ return node['right']
+
+# Classification and Regression Tree Algorithm
+def decision_tree(train, test, max_depth, min_size):
+ tree = build_tree(train, max_depth, min_size)
+ predictions = list()
+ for row in test:
+ prediction = predict(tree, row)
+ predictions.append(prediction)
+ return(predictions)
+
+# Test CART
+seed(1)
+# load and prepare data
+filename = 'DataFiles/rideclass.csv'
+dataset = load_csv(filename)
+# convert string attributes to integers
+for i in range(len(dataset[0])):
+ str_column_to_float(dataset, i)
+# evaluate algorithm
+n_folds = 5
+max_depth = 5
+min_size = 10
+scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
+print('Scores: %s' % scores)
+print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+
+
+
+
+
Entropy and the ID3 algorithm
ID3, learns decision trees by constructing
@@ -643,7 +830,7 @@ attributes at each step while growing the tree.
-
Implementing the ID3 Algorithm
+Implementing the ID3 Algorithm
more text to come here, material presented during lecture Friday Oct 25.
@@ -651,7 +838,7 @@ attributes at each step while growing the tree.
-
Cancer Data again now with Decision Trees
+Cancer Data again now with Decision Trees
@@ -700,7 +887,7 @@ deep_tree_clf.fit(X_train_scaled, y_train)
-
Another example, the moons again
+Another example, the moons again
@@ -772,7 +959,7 @@ plt.show()
-
Playing around with regions
+Playing around with regions
@@ -800,7 +987,7 @@ plt.show()
-
Regression trees
+Regression trees
@@ -822,7 +1009,7 @@ tree_reg.fit(X, y)
-
Final regressor code
+Final regressor code
@@ -900,7 +1087,7 @@ plt.show()
-
Pros and cons of trees, pros
+Pros and cons of trees, pros
- White box, easy to interpret model. Some people believe that decision trees more closely mirror human decision-making than do the regression and classification approaches discussed earlier (think of support vector machines)
@@ -914,7 +1101,7 @@ plt.show()
-Disadvantages
+Disadvantages
- Unfortunately, trees generally do not have the same level of predictive accuracy as some of the other regression and classification approaches
@@ -931,7 +1118,7 @@ However, by aggregating many decision trees, using methods like bagging, random
-
Bagging
+Bagging
The plain decision trees suffer from high
@@ -974,7 +1161,7 @@ predictor, averaged over all \( B \) trees.
-
Simple example, head or tail
+Simple example, head or tail
@@ -994,7 +1181,7 @@ plt.show()
-
Random forests
+Random forests
Random forests provide an improvement over bagged trees by way of a
@@ -1038,7 +1225,7 @@ this setting.
-
A simple scikit-learn example
+A simple scikit-learn example
@@ -1056,7 +1243,7 @@ accuracy = cross_validate(Random_Forest_mode
-
Please, not the moons again!
+Please, not the moons again!
@@ -1114,7 +1301,7 @@ voting_clf.fit(X_train, y_train)
-
Bagging examples
+Bagging examples
@@ -1175,7 +1362,7 @@ plt.show()
-
Then random forests
+Then random forests
@@ -1197,7 +1384,7 @@ np.sum(y_pred =
-
Boosting and more
+Boosting and more
More material to come here.
diff --git a/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb b/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb
index 3a926590f..ac3afe181 100644
--- a/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb
+++ b/doc/pub/DecisionTrees/ipynb/DecisionTrees.ipynb
@@ -10,7 +10,7 @@
" \n",
"**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University\n",
"\n",
- "Date: **Oct 26, 2019**\n",
+ "Date: **Oct 29, 2019**\n",
"\n",
"Copyright 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license\n",
"\n",
@@ -502,6 +502,196 @@
"$$"
]
},
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "## The CART (Classification and Regression Tree) algorithm\n",
+ "\n",
+ "The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 2,
+ "metadata": {
+ "collapsed": false
+ },
+ "outputs": [],
+ "source": [
+ "from random import seed\n",
+ "from random import randrange\n",
+ "from csv import reader\n",
+ " \n",
+ "# Load a CSV file\n",
+ "def load_csv(filename):\n",
+ "\tfile = open(filename, \"rb\")\n",
+ "\tlines = reader(file)\n",
+ "\tdataset = list(lines)\n",
+ "\treturn dataset\n",
+ " \n",
+ "# Convert string column to float\n",
+ "def str_column_to_float(dataset, column):\n",
+ "\tfor row in dataset:\n",
+ "\t\trow[column] = float(row[column].strip())\n",
+ " \n",
+ "# Split a dataset into k folds\n",
+ "def cross_validation_split(dataset, n_folds):\n",
+ "\tdataset_split = list()\n",
+ "\tdataset_copy = list(dataset)\n",
+ "\tfold_size = int(len(dataset) / n_folds)\n",
+ "\tfor i in range(n_folds):\n",
+ "\t\tfold = list()\n",
+ "\t\twhile len(fold) < fold_size:\n",
+ "\t\t\tindex = randrange(len(dataset_copy))\n",
+ "\t\t\tfold.append(dataset_copy.pop(index))\n",
+ "\t\tdataset_split.append(fold)\n",
+ "\treturn dataset_split\n",
+ " \n",
+ "# Calculate accuracy percentage\n",
+ "def accuracy_metric(actual, predicted):\n",
+ "\tcorrect = 0\n",
+ "\tfor i in range(len(actual)):\n",
+ "\t\tif actual[i] == predicted[i]:\n",
+ "\t\t\tcorrect += 1\n",
+ "\treturn correct / float(len(actual)) * 100.0\n",
+ " \n",
+ "# Evaluate an algorithm using a cross validation split\n",
+ "def evaluate_algorithm(dataset, algorithm, n_folds, *args):\n",
+ "\tfolds = cross_validation_split(dataset, n_folds)\n",
+ "\tscores = list()\n",
+ "\tfor fold in folds:\n",
+ "\t\ttrain_set = list(folds)\n",
+ "\t\ttrain_set.remove(fold)\n",
+ "\t\ttrain_set = sum(train_set, [])\n",
+ "\t\ttest_set = list()\n",
+ "\t\tfor row in fold:\n",
+ "\t\t\trow_copy = list(row)\n",
+ "\t\t\ttest_set.append(row_copy)\n",
+ "\t\t\trow_copy[-1] = None\n",
+ "\t\tpredicted = algorithm(train_set, test_set, *args)\n",
+ "\t\tactual = [row[-1] for row in fold]\n",
+ "\t\taccuracy = accuracy_metric(actual, predicted)\n",
+ "\t\tscores.append(accuracy)\n",
+ "\treturn scores\n",
+ " \n",
+ "# Split a dataset based on an attribute and an attribute value\n",
+ "def test_split(index, value, dataset):\n",
+ "\tleft, right = list(), list()\n",
+ "\tfor row in dataset:\n",
+ "\t\tif row[index] < value:\n",
+ "\t\t\tleft.append(row)\n",
+ "\t\telse:\n",
+ "\t\t\tright.append(row)\n",
+ "\treturn left, right\n",
+ " \n",
+ "# Calculate the Gini index for a split dataset\n",
+ "def gini_index(groups, classes):\n",
+ "\t# count all samples at split point\n",
+ "\tn_instances = float(sum([len(group) for group in groups]))\n",
+ "\t# sum weighted Gini index for each group\n",
+ "\tgini = 0.0\n",
+ "\tfor group in groups:\n",
+ "\t\tsize = float(len(group))\n",
+ "\t\t# avoid divide by zero\n",
+ "\t\tif size == 0:\n",
+ "\t\t\tcontinue\n",
+ "\t\tscore = 0.0\n",
+ "\t\t# score the group based on the score for each class\n",
+ "\t\tfor class_val in classes:\n",
+ "\t\t\tp = [row[-1] for row in group].count(class_val) / size\n",
+ "\t\t\tscore += p * p\n",
+ "\t\t# weight the group score by its relative size\n",
+ "\t\tgini += (1.0 - score) * (size / n_instances)\n",
+ "\treturn gini\n",
+ " \n",
+ "# Select the best split point for a dataset\n",
+ "def get_split(dataset):\n",
+ "\tclass_values = list(set(row[-1] for row in dataset))\n",
+ "\tb_index, b_value, b_score, b_groups = 999, 999, 999, None\n",
+ "\tfor index in range(len(dataset[0])-1):\n",
+ "\t\tfor row in dataset:\n",
+ "\t\t\tgroups = test_split(index, row[index], dataset)\n",
+ "\t\t\tgini = gini_index(groups, class_values)\n",
+ "\t\t\tif gini < b_score:\n",
+ "\t\t\t\tb_index, b_value, b_score, b_groups = index, row[index], gini, groups\n",
+ "\treturn {'index':b_index, 'value':b_value, 'groups':b_groups}\n",
+ " \n",
+ "# Create a terminal node value\n",
+ "def to_terminal(group):\n",
+ "\toutcomes = [row[-1] for row in group]\n",
+ "\treturn max(set(outcomes), key=outcomes.count)\n",
+ " \n",
+ "# Create child splits for a node or make terminal\n",
+ "def split(node, max_depth, min_size, depth):\n",
+ "\tleft, right = node['groups']\n",
+ "\tdel(node['groups'])\n",
+ "\t# check for a no split\n",
+ "\tif not left or not right:\n",
+ "\t\tnode['left'] = node['right'] = to_terminal(left + right)\n",
+ "\t\treturn\n",
+ "\t# check for max depth\n",
+ "\tif depth >= max_depth:\n",
+ "\t\tnode['left'], node['right'] = to_terminal(left), to_terminal(right)\n",
+ "\t\treturn\n",
+ "\t# process left child\n",
+ "\tif len(left) <= min_size:\n",
+ "\t\tnode['left'] = to_terminal(left)\n",
+ "\telse:\n",
+ "\t\tnode['left'] = get_split(left)\n",
+ "\t\tsplit(node['left'], max_depth, min_size, depth+1)\n",
+ "\t# process right child\n",
+ "\tif len(right) <= min_size:\n",
+ "\t\tnode['right'] = to_terminal(right)\n",
+ "\telse:\n",
+ "\t\tnode['right'] = get_split(right)\n",
+ "\t\tsplit(node['right'], max_depth, min_size, depth+1)\n",
+ " \n",
+ "# Build a decision tree\n",
+ "def build_tree(train, max_depth, min_size):\n",
+ "\troot = get_split(train)\n",
+ "\tsplit(root, max_depth, min_size, 1)\n",
+ "\treturn root\n",
+ " \n",
+ "# Make a prediction with a decision tree\n",
+ "def predict(node, row):\n",
+ "\tif row[node['index']] < node['value']:\n",
+ "\t\tif isinstance(node['left'], dict):\n",
+ "\t\t\treturn predict(node['left'], row)\n",
+ "\t\telse:\n",
+ "\t\t\treturn node['left']\n",
+ "\telse:\n",
+ "\t\tif isinstance(node['right'], dict):\n",
+ "\t\t\treturn predict(node['right'], row)\n",
+ "\t\telse:\n",
+ "\t\t\treturn node['right']\n",
+ " \n",
+ "# Classification and Regression Tree Algorithm\n",
+ "def decision_tree(train, test, max_depth, min_size):\n",
+ "\ttree = build_tree(train, max_depth, min_size)\n",
+ "\tpredictions = list()\n",
+ "\tfor row in test:\n",
+ "\t\tprediction = predict(tree, row)\n",
+ "\t\tpredictions.append(prediction)\n",
+ "\treturn(predictions)\n",
+ " \n",
+ "# Test CART \n",
+ "seed(1)\n",
+ "# load and prepare data\n",
+ "filename = 'DataFiles/rideclass.csv'\n",
+ "dataset = load_csv(filename)\n",
+ "# convert string attributes to integers\n",
+ "for i in range(len(dataset[0])):\n",
+ "\tstr_column_to_float(dataset, i)\n",
+ "# evaluate algorithm\n",
+ "n_folds = 5\n",
+ "max_depth = 5\n",
+ "min_size = 10\n",
+ "scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)\n",
+ "print('Scores: %s' % scores)\n",
+ "print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))"
+ ]
+ },
{
"cell_type": "markdown",
"metadata": {},
@@ -546,7 +736,7 @@
},
{
"cell_type": "code",
- "execution_count": 2,
+ "execution_count": 3,
"metadata": {
"collapsed": false
},
@@ -604,7 +794,7 @@
},
{
"cell_type": "code",
- "execution_count": 3,
+ "execution_count": 4,
"metadata": {
"collapsed": false
},
@@ -685,7 +875,7 @@
},
{
"cell_type": "code",
- "execution_count": 4,
+ "execution_count": 5,
"metadata": {
"collapsed": false
},
@@ -722,7 +912,7 @@
},
{
"cell_type": "code",
- "execution_count": 5,
+ "execution_count": 6,
"metadata": {
"collapsed": false
},
@@ -738,7 +928,7 @@
},
{
"cell_type": "code",
- "execution_count": 6,
+ "execution_count": 7,
"metadata": {
"collapsed": false
},
@@ -759,7 +949,7 @@
},
{
"cell_type": "code",
- "execution_count": 7,
+ "execution_count": 8,
"metadata": {
"collapsed": false
},
@@ -807,7 +997,7 @@
},
{
"cell_type": "code",
- "execution_count": 8,
+ "execution_count": 9,
"metadata": {
"collapsed": false
},
@@ -923,7 +1113,7 @@
},
{
"cell_type": "code",
- "execution_count": 9,
+ "execution_count": 10,
"metadata": {
"collapsed": false
},
@@ -998,7 +1188,7 @@
},
{
"cell_type": "code",
- "execution_count": 10,
+ "execution_count": 11,
"metadata": {
"collapsed": false
},
@@ -1025,7 +1215,7 @@
},
{
"cell_type": "code",
- "execution_count": 11,
+ "execution_count": 12,
"metadata": {
"collapsed": false
},
@@ -1053,7 +1243,7 @@
},
{
"cell_type": "code",
- "execution_count": 12,
+ "execution_count": 13,
"metadata": {
"collapsed": false
},
@@ -1069,7 +1259,7 @@
},
{
"cell_type": "code",
- "execution_count": 13,
+ "execution_count": 14,
"metadata": {
"collapsed": false
},
@@ -1087,7 +1277,7 @@
},
{
"cell_type": "code",
- "execution_count": 14,
+ "execution_count": 15,
"metadata": {
"collapsed": false
},
@@ -1110,7 +1300,7 @@
},
{
"cell_type": "code",
- "execution_count": 15,
+ "execution_count": 16,
"metadata": {
"collapsed": false
},
@@ -1128,7 +1318,7 @@
},
{
"cell_type": "code",
- "execution_count": 16,
+ "execution_count": 17,
"metadata": {
"collapsed": false
},
@@ -1140,7 +1330,7 @@
},
{
"cell_type": "code",
- "execution_count": 17,
+ "execution_count": 18,
"metadata": {
"collapsed": false
},
@@ -1154,7 +1344,7 @@
},
{
"cell_type": "code",
- "execution_count": 18,
+ "execution_count": 19,
"metadata": {
"collapsed": false
},
@@ -1197,7 +1387,7 @@
},
{
"cell_type": "code",
- "execution_count": 19,
+ "execution_count": 20,
"metadata": {
"collapsed": false
},
@@ -1210,7 +1400,7 @@
},
{
"cell_type": "code",
- "execution_count": 20,
+ "execution_count": 21,
"metadata": {
"collapsed": false
},
diff --git a/doc/pub/DecisionTrees/ipynb/ipynb-DecisionTrees-src.tar.gz b/doc/pub/DecisionTrees/ipynb/ipynb-DecisionTrees-src.tar.gz
index d19892834..88c12c16b 100644
Binary files a/doc/pub/DecisionTrees/ipynb/ipynb-DecisionTrees-src.tar.gz and b/doc/pub/DecisionTrees/ipynb/ipynb-DecisionTrees-src.tar.gz differ
diff --git a/doc/pub/DecisionTrees/pdf/DecisionTrees-minted.pdf b/doc/pub/DecisionTrees/pdf/DecisionTrees-minted.pdf
index 9f633fb4e..4f7eedff7 100644
Binary files a/doc/pub/DecisionTrees/pdf/DecisionTrees-minted.pdf and b/doc/pub/DecisionTrees/pdf/DecisionTrees-minted.pdf differ
diff --git a/doc/src/DecisionTrees/DecisionTrees.do.txt b/doc/src/DecisionTrees/DecisionTrees.do.txt
index e0bbfb497..44b510e89 100644
--- a/doc/src/DecisionTrees/DecisionTrees.do.txt
+++ b/doc/src/DecisionTrees/DecisionTrees.do.txt
@@ -399,6 +399,186 @@ s = -\sum_{k=1}^K p_{mk}\log{p_{mk}}.
!et
+!split
+===== The CART (Classification and Regression Tree) algorithm =====
+
+The above functions (gini, entropy and misclassification error) are important components of the so-called CART algorithm. We will discuss this algorithm first before we move on to the information gain algorithm ID3.
+
+!bc pycod
+from random import seed
+from random import randrange
+from csv import reader
+
+# Load a CSV file
+def load_csv(filename):
+ file = open(filename, "rb")
+ lines = reader(file)
+ dataset = list(lines)
+ return dataset
+
+# Convert string column to float
+def str_column_to_float(dataset, column):
+ for row in dataset:
+ row[column] = float(row[column].strip())
+
+# Split a dataset into k folds
+def cross_validation_split(dataset, n_folds):
+ dataset_split = list()
+ dataset_copy = list(dataset)
+ fold_size = int(len(dataset) / n_folds)
+ for i in range(n_folds):
+ fold = list()
+ while len(fold) < fold_size:
+ index = randrange(len(dataset_copy))
+ fold.append(dataset_copy.pop(index))
+ dataset_split.append(fold)
+ return dataset_split
+
+# Calculate accuracy percentage
+def accuracy_metric(actual, predicted):
+ correct = 0
+ for i in range(len(actual)):
+ if actual[i] == predicted[i]:
+ correct += 1
+ return correct / float(len(actual)) * 100.0
+
+# Evaluate an algorithm using a cross validation split
+def evaluate_algorithm(dataset, algorithm, n_folds, *args):
+ folds = cross_validation_split(dataset, n_folds)
+ scores = list()
+ for fold in folds:
+ train_set = list(folds)
+ train_set.remove(fold)
+ train_set = sum(train_set, [])
+ test_set = list()
+ for row in fold:
+ row_copy = list(row)
+ test_set.append(row_copy)
+ row_copy[-1] = None
+ predicted = algorithm(train_set, test_set, *args)
+ actual = [row[-1] for row in fold]
+ accuracy = accuracy_metric(actual, predicted)
+ scores.append(accuracy)
+ return scores
+
+# Split a dataset based on an attribute and an attribute value
+def test_split(index, value, dataset):
+ left, right = list(), list()
+ for row in dataset:
+ if row[index] < value:
+ left.append(row)
+ else:
+ right.append(row)
+ return left, right
+
+# Calculate the Gini index for a split dataset
+def gini_index(groups, classes):
+ # count all samples at split point
+ n_instances = float(sum([len(group) for group in groups]))
+ # sum weighted Gini index for each group
+ gini = 0.0
+ for group in groups:
+ size = float(len(group))
+ # avoid divide by zero
+ if size == 0:
+ continue
+ score = 0.0
+ # score the group based on the score for each class
+ for class_val in classes:
+ p = [row[-1] for row in group].count(class_val) / size
+ score += p * p
+ # weight the group score by its relative size
+ gini += (1.0 - score) * (size / n_instances)
+ return gini
+
+# Select the best split point for a dataset
+def get_split(dataset):
+ class_values = list(set(row[-1] for row in dataset))
+ b_index, b_value, b_score, b_groups = 999, 999, 999, None
+ for index in range(len(dataset[0])-1):
+ for row in dataset:
+ groups = test_split(index, row[index], dataset)
+ gini = gini_index(groups, class_values)
+ if gini < b_score:
+ b_index, b_value, b_score, b_groups = index, row[index], gini, groups
+ return {'index':b_index, 'value':b_value, 'groups':b_groups}
+
+# Create a terminal node value
+def to_terminal(group):
+ outcomes = [row[-1] for row in group]
+ return max(set(outcomes), key=outcomes.count)
+
+# Create child splits for a node or make terminal
+def split(node, max_depth, min_size, depth):
+ left, right = node['groups']
+ del(node['groups'])
+ # check for a no split
+ if not left or not right:
+ node['left'] = node['right'] = to_terminal(left + right)
+ return
+ # check for max depth
+ if depth >= max_depth:
+ node['left'], node['right'] = to_terminal(left), to_terminal(right)
+ return
+ # process left child
+ if len(left) <= min_size:
+ node['left'] = to_terminal(left)
+ else:
+ node['left'] = get_split(left)
+ split(node['left'], max_depth, min_size, depth+1)
+ # process right child
+ if len(right) <= min_size:
+ node['right'] = to_terminal(right)
+ else:
+ node['right'] = get_split(right)
+ split(node['right'], max_depth, min_size, depth+1)
+
+# Build a decision tree
+def build_tree(train, max_depth, min_size):
+ root = get_split(train)
+ split(root, max_depth, min_size, 1)
+ return root
+
+# Make a prediction with a decision tree
+def predict(node, row):
+ if row[node['index']] < node['value']:
+ if isinstance(node['left'], dict):
+ return predict(node['left'], row)
+ else:
+ return node['left']
+ else:
+ if isinstance(node['right'], dict):
+ return predict(node['right'], row)
+ else:
+ return node['right']
+
+# Classification and Regression Tree Algorithm
+def decision_tree(train, test, max_depth, min_size):
+ tree = build_tree(train, max_depth, min_size)
+ predictions = list()
+ for row in test:
+ prediction = predict(tree, row)
+ predictions.append(prediction)
+ return(predictions)
+
+# Test CART
+seed(1)
+# load and prepare data
+filename = 'DataFiles/rideclass.csv'
+dataset = load_csv(filename)
+# convert string attributes to integers
+for i in range(len(dataset[0])):
+ str_column_to_float(dataset, i)
+# evaluate algorithm
+n_folds = 5
+max_depth = 5
+min_size = 10
+scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
+print('Scores: %s' % scores)
+print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
+
+!ec
+
!split
===== Entropy and the ID3 algorithm =====
diff --git a/doc/src/DecisionTrees/cart.py~ b/doc/src/DecisionTrees/cart.py~
deleted file mode 100644
index ee6c049b7..000000000
--- a/doc/src/DecisionTrees/cart.py~
+++ /dev/null
@@ -1,172 +0,0 @@
-# CART on the Bank Note dataset
-from random import seed
-from random import randrange
-from csv import reader
-
-# Load a CSV file
-def load_csv(filename):
- file = open(filename, "rb")
- lines = reader(file)
- dataset = list(lines)
- return dataset
-
-# Convert string column to float
-def str_column_to_float(dataset, column):
- for row in dataset:
- row[column] = float(row[column].strip())
-
-# Split a dataset into k folds
-def cross_validation_split(dataset, n_folds):
- dataset_split = list()
- dataset_copy = list(dataset)
- fold_size = int(len(dataset) / n_folds)
- for i in range(n_folds):
- fold = list()
- while len(fold) < fold_size:
- index = randrange(len(dataset_copy))
- fold.append(dataset_copy.pop(index))
- dataset_split.append(fold)
- return dataset_split
-
-# Calculate accuracy percentage
-def accuracy_metric(actual, predicted):
- correct = 0
- for i in range(len(actual)):
- if actual[i] == predicted[i]:
- correct += 1
- return correct / float(len(actual)) * 100.0
-
-# Evaluate an algorithm using a cross validation split
-def evaluate_algorithm(dataset, algorithm, n_folds, *args):
- folds = cross_validation_split(dataset, n_folds)
- scores = list()
- for fold in folds:
- train_set = list(folds)
- train_set.remove(fold)
- train_set = sum(train_set, [])
- test_set = list()
- for row in fold:
- row_copy = list(row)
- test_set.append(row_copy)
- row_copy[-1] = None
- predicted = algorithm(train_set, test_set, *args)
- actual = [row[-1] for row in fold]
- accuracy = accuracy_metric(actual, predicted)
- scores.append(accuracy)
- return scores
-
-# Split a dataset based on an attribute and an attribute value
-def test_split(index, value, dataset):
- left, right = list(), list()
- for row in dataset:
- if row[index] < value:
- left.append(row)
- else:
- right.append(row)
- return left, right
-
-# Calculate the Gini index for a split dataset
-def gini_index(groups, classes):
- # count all samples at split point
- n_instances = float(sum([len(group) for group in groups]))
- # sum weighted Gini index for each group
- gini = 0.0
- for group in groups:
- size = float(len(group))
- # avoid divide by zero
- if size == 0:
- continue
- score = 0.0
- # score the group based on the score for each class
- for class_val in classes:
- p = [row[-1] for row in group].count(class_val) / size
- score += p * p
- # weight the group score by its relative size
- gini += (1.0 - score) * (size / n_instances)
- return gini
-
-# Select the best split point for a dataset
-def get_split(dataset):
- class_values = list(set(row[-1] for row in dataset))
- b_index, b_value, b_score, b_groups = 999, 999, 999, None
- for index in range(len(dataset[0])-1):
- for row in dataset:
- groups = test_split(index, row[index], dataset)
- gini = gini_index(groups, class_values)
- if gini < b_score:
- b_index, b_value, b_score, b_groups = index, row[index], gini, groups
- return {'index':b_index, 'value':b_value, 'groups':b_groups}
-
-# Create a terminal node value
-def to_terminal(group):
- outcomes = [row[-1] for row in group]
- return max(set(outcomes), key=outcomes.count)
-
-# Create child splits for a node or make terminal
-def split(node, max_depth, min_size, depth):
- left, right = node['groups']
- del(node['groups'])
- # check for a no split
- if not left or not right:
- node['left'] = node['right'] = to_terminal(left + right)
- return
- # check for max depth
- if depth >= max_depth:
- node['left'], node['right'] = to_terminal(left), to_terminal(right)
- return
- # process left child
- if len(left) <= min_size:
- node['left'] = to_terminal(left)
- else:
- node['left'] = get_split(left)
- split(node['left'], max_depth, min_size, depth+1)
- # process right child
- if len(right) <= min_size:
- node['right'] = to_terminal(right)
- else:
- node['right'] = get_split(right)
- split(node['right'], max_depth, min_size, depth+1)
-
-# Build a decision tree
-def build_tree(train, max_depth, min_size):
- root = get_split(train)
- split(root, max_depth, min_size, 1)
- return root
-
-# Make a prediction with a decision tree
-def predict(node, row):
- if row[node['index']] < node['value']:
- if isinstance(node['left'], dict):
- return predict(node['left'], row)
- else:
- return node['left']
- else:
- if isinstance(node['right'], dict):
- return predict(node['right'], row)
- else:
- return node['right']
-
-# Classification and Regression Tree Algorithm
-def decision_tree(train, test, max_depth, min_size):
- tree = build_tree(train, max_depth, min_size)
- predictions = list()
- for row in test:
- prediction = predict(tree, row)
- predictions.append(prediction)
- return(predictions)
-
-# Test CART on Bank Note dataset
-seed(1)
-# load and prepare data
-filename = 'DataFiles/ride.csv'
-dataset = load_csv(filename)
-# convert string attributes to integers
-for i in range(len(dataset[0])):
- str_column_to_float(dataset, i)
-# evaluate algorithm
-n_folds = 5
-max_depth = 5
-min_size = 10
-scores = evaluate_algorithm(dataset, decision_tree, n_folds, max_depth, min_size)
-print('Scores: %s' % scores)
-print('Mean Accuracy: %.3f%%' % (sum(scores)/float(len(scores))))
diff --git a/doc/src/DecisionTrees/decisiontree.py~ b/doc/src/DecisionTrees/decisiontree.py~
deleted file mode 100644
index 8b3420f3e..000000000
--- a/doc/src/DecisionTrees/decisiontree.py~
+++ /dev/null
@@ -1,187 +0,0 @@
-import re
-import math
-from collections import deque
-
-# x is examples in training set
-# y is set of attributes
-# label is target attributes
-# Node is a class which has properties values, childs, and next
-# root is top node in the decision tree
-
-class Node(object):
- def __init__(self):
- self.value = None
- self.next = None
- self.childs = None
-
-# Simple class of Decision Tree
-# Aimed for who want to learn Decision Tree, so it is not optimized
-class DecisionTree(object):
- def __init__(self, sample, attributes, labels):
- self.sample = sample
- self.attributes = attributes
- self.labels = labels
- self.labelCodes = None
- self.labelCodesCount = None
- self.initLabelCodes()
- # print(self.labelCodes)
- self.root = None
- self.entropy = self.getEntropy([x for x in range(len(self.labels))])
-
- def initLabelCodes(self):
- self.labelCodes = []
- self.labelCodesCount = []
- for l in self.labels:
- if l not in self.labelCodes:
- self.labelCodes.append(l)
- self.labelCodesCount.append(0)
- self.labelCodesCount[self.labelCodes.index(l)] += 1
-
- def getLabelCodeId(self, sampleId):
- return self.labelCodes.index(self.labels[sampleId])
-
- def getAttributeValues(self, sampleIds, attributeId):
- vals = []
- for sid in sampleIds:
- val = self.sample[sid][attributeId]
- if val not in vals:
- vals.append(val)
- # print(vals)
- return vals
-
- def getEntropy(self, sampleIds):
- entropy = 0
- labelCount = [0] * len(self.labelCodes)
- for sid in sampleIds:
- labelCount[self.getLabelCodeId(sid)] += 1
- # print("-ge", labelCount)
- for lv in labelCount:
- # print(lv)
- if lv != 0:
- entropy += -lv/len(sampleIds) * math.log(lv/len(sampleIds), 2)
- else:
- entropy += 0
- return entropy
-
- def getDominantLabel(self, sampleIds):
- labelCodesCount = [0] * len(self.labelCodes)
- for sid in sampleIds:
- labelCodesCount[self.labelCodes.index(self.labels[sid])] += 1
- return self.labelCodes[labelCodesCount.index(max(labelCodesCount))]
-
- def getInformationGain(self, sampleIds, attributeId):
- gain = self.getEntropy(sampleIds)
- attributeVals = []
- attributeValsCount = []
- attributeValsIds = []
- for sid in sampleIds:
- val = self.sample[sid][attributeId]
- if val not in attributeVals:
- attributeVals.append(val)
- attributeValsCount.append(0)
- attributeValsIds.append([])
- vid = attributeVals.index(val)
- attributeValsCount[vid] += 1
- attributeValsIds[vid].append(sid)
- # print("-gig", self.attributes[attributeId])
- for vc, vids in zip(attributeValsCount, attributeValsIds):
- # print("-gig", vids)
- gain -= vc/len(sampleIds) * self.getEntropy(vids)
- return gain
-
- def getAttributeMaxInformationGain(self, sampleIds, attributeIds):
- attributesEntropy = [0] * len(attributeIds)
- for i, attId in zip(range(len(attributeIds)), attributeIds):
- attributesEntropy[i] = self.getInformationGain(sampleIds, attId)
- maxId = attributeIds[attributesEntropy.index(max(attributesEntropy))]
- return self.attributes[maxId], maxId
-
- def isSingleLabeled(self, sampleIds):
- label = self.labels[sampleIds[0]]
- for sid in sampleIds:
- if self.labels[sid] != label:
- return False
- return True
-
- def getLabel(self, sampleId):
- return self.labels[sampleId]
-
- def id3(self):
- sampleIds = [x for x in range(len(self.sample))]
- attributeIds = [x for x in range(len(self.attributes))]
- self.root = self.id3Recv(sampleIds, attributeIds, self.root)
-
- def id3Recv(self, sampleIds, attributeIds, root):
- root = Node() # Initialize current root
- if self.isSingleLabeled(sampleIds):
- root.value = self.labels[sampleIds[0]]
- return root
- # print(attributeIds)
- if len(attributeIds) == 0:
- root.value = self.getDominantLabel(sampleIds)
- return root
- bestAttrName, bestAttrId = self.getAttributeMaxInformationGain(
- sampleIds, attributeIds)
- # print(bestAttrName)
- root.value = bestAttrName
- root.childs = [] # Create list of children
- for value in self.getAttributeValues(sampleIds, bestAttrId):
- # print(value)
- child = Node()
- child.value = value
- root.childs.append(child) # Append new child node to current
- # root
- childSampleIds = []
- for sid in sampleIds:
- if self.sample[sid][bestAttrId] == value:
- childSampleIds.append(sid)
- if len(childSampleIds) == 0:
- child.next = self.getDominantLabel(sampleIds)
- else:
- # print(bestAttrName, bestAttrId)
- # print(attributeIds)
- if len(attributeIds) > 0 and bestAttrId in attributeIds:
- toRemove = attributeIds.index(bestAttrId)
- attributeIds.pop(toRemove)
- child.next = self.id3Recv(
- childSampleIds, attributeIds, child.next)
- return root
-
- def printTree(self):
- if self.root:
- roots = deque()
- roots.append(self.root)
- while len(roots) > 0:
- root = roots.popleft()
- print(root.value)
- if root.childs:
- for child in root.childs:
- print('({})'.format(child.value))
- roots.append(child.next)
- elif root.next:
- print(root.next)
-
-
-def test():
- f = open('rideclass.csv')
- attributes = f.readline().split(',')
- attributes = attributes[1:len(attributes)-1]
- print(attributes)
- sample = f.readlines()
- f.close()
- for i in range(len(sample)):
- sample[i] = re.sub('\d+,', '', sample[i])
- sample[i] = sample[i].strip().split(',')
- labels = []
- for s in sample:
- labels.append(s.pop())
- # print(sample)
- # print(labels)
- decisionTree = DecisionTree(sample, attributes, labels)
- print("System entropy {}".format(decisionTree.entropy))
- decisionTree.id3()
- decisionTree.printTree()
-
-
-if __name__ == '__main__':
- test()
diff --git a/doc/src/DecisionTrees/dtcancer.py~ b/doc/src/DecisionTrees/dtcancer.py~
deleted file mode 100644
index 0008c1d32..000000000
--- a/doc/src/DecisionTrees/dtcancer.py~
+++ /dev/null
@@ -1,29 +0,0 @@
-from sklearn.datasets import load_breast_cancer
-from sklearn.tree import DecisionTreeClassifier
-from sklearn.model_selection import train_test_split
-from sklearn.metrics import confusion_matrix
-from sklearn.tree import export_graphviz
-
-from IPython.display import Image
-from pydot import graph_from_dot_data
-import pandas as pd
-import numpy as np
-
-cancer = load_breast_cancer()
-X = pd.DataFrame(cancer.data, columns=cancer.feature_names)
-y = pd.Categorical.from_codes(cancer.target, cancer.target_names)
-y = pd.get_dummies(y)
-
-X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1)
-tree_clf = DecisionTreeClassifier(max_depth=5)
-tree_clf.fit(X_train, y_train)
-
-export_graphviz(
- tree_clf,
- out_file="cancer.dot",
- feature_names=cancer.feature_names,
- class_names=cancer.target_names,
- rounded=True,
- filled=True
-)
-
diff --git a/doc/src/DecisionTrees/read.py~ b/doc/src/DecisionTrees/read.py~
deleted file mode 100644
index cc7dadd8b..000000000
--- a/doc/src/DecisionTrees/read.py~
+++ /dev/null
@@ -1,80 +0,0 @@
-# Common imports
-import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.tree import DecisionTreeClassifier
-from sklearn.model_selection import train_test_split
-from sklearn.tree import export_graphviz
-from sklearn.preprocessing import StandardScaler, OneHotEncoder
-from sklearn.compose import ColumnTransformer
-from IPython.display import Image
-from pydot import graph_from_dot_data
-import os
-
-# Where to save the figures and data files
-PROJECT_ROOT_DIR = "Results"
-FIGURE_ID = "Results/FigureFiles"
-DATA_ID = "DataFiles/"
-
-if not os.path.exists(PROJECT_ROOT_DIR):
- os.mkdir(PROJECT_ROOT_DIR)
-
-if not os.path.exists(FIGURE_ID):
- os.makedirs(FIGURE_ID)
-
-if not os.path.exists(DATA_ID):
- os.makedirs(DATA_ID)
-
-def image_path(fig_id):
- return os.path.join(FIGURE_ID, fig_id)
-
-def data_path(dat_id):
- return os.path.join(DATA_ID, dat_id)
-
-def save_fig(fig_id):
- plt.savefig(image_path(fig_id) + ".png", format='png')
-
-infile = open(data_path("ride.csv"),'r')
-
-# Read the experimental data with Pandas
-from IPython.display import display
-ridedata = pd.read_csv(infile,names = ('Outlook','Temperature','Humidity','Wind','Ride'))
-ridedata = pd.DataFrame(ridedata)
-display(ridedata)
-# Features and targets
-X = ridedata.loc[:, ridedata.columns != 'Ride'].values
-display(X)
-y = ridedata.loc[:, ridedata.columns == 'Ride'].values
-display(y)
-# Categorical variables to one-hot's
-onehotencoder = OneHotEncoder(categories="auto")
-
-X = ColumnTransformer(
- [("", onehotencoder)],
- remainder="passthrough").fit_transform(X)
-y.shape
-
-display(X)
-display(y)
-
-
-"""
-X = pd.DataFrame(ridedata.data, columns=ridedata.feature_names)
-y = pd.Categorical.from_codes(ridedata.target, ridedata.target_names)
-y = pd.get_dummies(y)
-
-
-tree_clf = DecisionTreeClassifier(max_depth=2)
-tree_clf.fit(X, y)
-
-
-export_graphviz(
- tree_clf,
- out_file="ride.dot",
- feature_names=tree_clf.feature_names,
- class_names=tree_clf.target_names,
- rounded=True,
- filled=True
-)
-"""
-