typos here and there
This commit is contained in:
+119
-302
@@ -1062,7 +1062,7 @@ print('Split: [X%d < %.3f]' % ((split['index']+1), split['value']))
|
||||
|
||||
|
||||
!split
|
||||
===== Another example, the moons again =====
|
||||
===== Another example, the moons =====
|
||||
!bc pycod
|
||||
from __future__ import division, print_function, unicode_literals
|
||||
|
||||
@@ -1306,300 +1306,6 @@ We discuss these methods here.
|
||||
|
||||
FIGURE: [DataFiles/ensembleoverview.png, width=600 frac=0.8]
|
||||
|
||||
|
||||
|
||||
!split
|
||||
===== Bagging =====
|
||||
|
||||
The _plain_ decision trees suffer from high
|
||||
variance. This means that if we split the training data into two parts
|
||||
at random, and fit a decision tree to both halves, the results that we
|
||||
get could be quite different. In contrast, a procedure with low
|
||||
variance will yield similar results if applied repeatedly to distinct
|
||||
data sets; linear regression tends to have low variance, if the ratio
|
||||
of $n$ to $p$ is moderately large.
|
||||
|
||||
_Bootstrap aggregation_, or just _bagging_, is a
|
||||
general-purpose procedure for reducing the variance of a statistical
|
||||
learning method.
|
||||
|
||||
|
||||
!split
|
||||
===== More bagging =====
|
||||
|
||||
Bagging typically results in improved accuracy
|
||||
over prediction using a single tree. Unfortunately, however, it can be
|
||||
difficult to interpret the resulting model. Recall that one of the
|
||||
advantages of decision trees is the attractive and easily interpreted
|
||||
diagram that results.
|
||||
|
||||
However, when we bag a large number of trees, it is no longer
|
||||
possible to represent the resulting statistical learning procedure
|
||||
using a single tree, and it is no longer clear which variables are
|
||||
most important to the procedure. Thus, bagging improves prediction
|
||||
accuracy at the expense of interpretability. Although the collection
|
||||
of bagged trees is much more difficult to interpret than a single
|
||||
tree, one can obtain an overall summary of the importance of each
|
||||
predictor using the MSE (for bagging regression trees) or the Gini
|
||||
index (for bagging classification trees). In the case of bagging
|
||||
regression trees, we can record the total amount that the MSE is
|
||||
decreased due to splits over a given predictor, averaged over all $B$ possible
|
||||
trees. A large value indicates an important predictor. Similarly, in
|
||||
the context of bagging classification trees, we can add up the total
|
||||
amount that the Gini index is decreased by splits over a given
|
||||
predictor, averaged over all $B$ trees.
|
||||
|
||||
!split
|
||||
===== Simple Voting Example, head or tail =====
|
||||
!bc pycod
|
||||
heads_proba = 0.51
|
||||
coin_tosses = (np.random.rand(10000, 10) < heads_proba).astype(np.int32)
|
||||
cumulative_heads_ratio = np.cumsum(coin_tosses, axis=0) / np.arange(1, 10001).reshape(-1, 1)
|
||||
plt.figure(figsize=(8,3.5))
|
||||
plt.plot(cumulative_heads_ratio)
|
||||
plt.plot([0, 10000], [0.51, 0.51], "k--", linewidth=2, label="51%")
|
||||
plt.plot([0, 10000], [0.5, 0.5], "k-", label="50%")
|
||||
plt.xlabel("Number of coin tosses")
|
||||
plt.ylabel("Heads ratio")
|
||||
plt.legend(loc="lower right")
|
||||
plt.axis([0, 10000, 0.42, 0.58])
|
||||
save_fig("votingsimple")
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== Using the Voting Classifier =====
|
||||
!bc pycod
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.datasets import make_moons
|
||||
|
||||
X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
|
||||
|
||||
from sklearn.ensemble import RandomForestClassifier
|
||||
from sklearn.ensemble import VotingClassifier
|
||||
from sklearn.linear_model import LogisticRegression
|
||||
from sklearn.svm import SVC
|
||||
|
||||
log_clf = LogisticRegression(solver="liblinear", random_state=42)
|
||||
rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
|
||||
svm_clf = SVC(gamma="auto", random_state=42)
|
||||
|
||||
voting_clf = VotingClassifier(
|
||||
estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
|
||||
voting='hard')
|
||||
|
||||
voting_clf.fit(X_train, y_train)
|
||||
|
||||
from sklearn.metrics import accuracy_score
|
||||
|
||||
for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
|
||||
clf.fit(X_train, y_train)
|
||||
y_pred = clf.predict(X_test)
|
||||
print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
|
||||
|
||||
log_clf = LogisticRegression(solver="liblinear", random_state=42)
|
||||
rnd_clf = RandomForestClassifier(n_estimators=10, random_state=42)
|
||||
svm_clf = SVC(gamma="auto", probability=True, random_state=42)
|
||||
|
||||
voting_clf = VotingClassifier(
|
||||
estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
|
||||
voting='soft')
|
||||
voting_clf.fit(X_train, y_train)
|
||||
|
||||
from sklearn.metrics import accuracy_score
|
||||
|
||||
for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
|
||||
clf.fit(X_train, y_train)
|
||||
y_pred = clf.predict(X_test)
|
||||
print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
|
||||
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== Please, not the moons again! Voting and Bagging =====
|
||||
|
||||
!bc pycod
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.datasets import make_moons
|
||||
|
||||
X, y = make_moons(n_samples=500, noise=0.30, random_state=42)
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42)
|
||||
from sklearn.ensemble import RandomForestClassifier
|
||||
from sklearn.ensemble import VotingClassifier
|
||||
from sklearn.linear_model import LogisticRegression
|
||||
from sklearn.svm import SVC
|
||||
|
||||
log_clf = LogisticRegression(random_state=42)
|
||||
rnd_clf = RandomForestClassifier(random_state=42)
|
||||
svm_clf = SVC(random_state=42)
|
||||
|
||||
voting_clf = VotingClassifier(
|
||||
estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
|
||||
voting='hard')
|
||||
voting_clf.fit(X_train, y_train)
|
||||
!ec
|
||||
|
||||
!bc pycod
|
||||
from sklearn.metrics import accuracy_score
|
||||
|
||||
for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
|
||||
clf.fit(X_train, y_train)
|
||||
y_pred = clf.predict(X_test)
|
||||
print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
|
||||
!ec
|
||||
|
||||
!bc pycod
|
||||
log_clf = LogisticRegression(random_state=42)
|
||||
rnd_clf = RandomForestClassifier(random_state=42)
|
||||
svm_clf = SVC(probability=True, random_state=42)
|
||||
|
||||
voting_clf = VotingClassifier(
|
||||
estimators=[('lr', log_clf), ('rf', rnd_clf), ('svc', svm_clf)],
|
||||
voting='soft')
|
||||
voting_clf.fit(X_train, y_train)
|
||||
!ec
|
||||
|
||||
!bc pycod
|
||||
from sklearn.metrics import accuracy_score
|
||||
|
||||
for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
|
||||
clf.fit(X_train, y_train)
|
||||
y_pred = clf.predict(X_test)
|
||||
print(clf.__class__.__name__, accuracy_score(y_test, y_pred))
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== Bagging Examples =====
|
||||
|
||||
!bc pycod
|
||||
from sklearn.ensemble import BaggingClassifier
|
||||
from sklearn.tree import DecisionTreeClassifier
|
||||
|
||||
bag_clf = BaggingClassifier(
|
||||
DecisionTreeClassifier(random_state=42), n_estimators=500,
|
||||
max_samples=100, bootstrap=True, n_jobs=-1, random_state=42)
|
||||
bag_clf.fit(X_train, y_train)
|
||||
y_pred = bag_clf.predict(X_test)
|
||||
!ec
|
||||
|
||||
|
||||
!bc pycod
|
||||
from sklearn.metrics import accuracy_score
|
||||
print(accuracy_score(y_test, y_pred))
|
||||
!ec
|
||||
|
||||
!bc pycod
|
||||
tree_clf = DecisionTreeClassifier(random_state=42)
|
||||
tree_clf.fit(X_train, y_train)
|
||||
y_pred_tree = tree_clf.predict(X_test)
|
||||
print(accuracy_score(y_test, y_pred_tree))
|
||||
!ec
|
||||
|
||||
!bc pycod
|
||||
from matplotlib.colors import ListedColormap
|
||||
|
||||
def plot_decision_boundary(clf, X, y, axes=[-1.5, 2.5, -1, 1.5], alpha=0.5, contour=True):
|
||||
x1s = np.linspace(axes[0], axes[1], 100)
|
||||
x2s = np.linspace(axes[2], axes[3], 100)
|
||||
x1, x2 = np.meshgrid(x1s, x2s)
|
||||
X_new = np.c_[x1.ravel(), x2.ravel()]
|
||||
y_pred = clf.predict(X_new).reshape(x1.shape)
|
||||
custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0'])
|
||||
plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap)
|
||||
if contour:
|
||||
custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50'])
|
||||
plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8)
|
||||
plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", alpha=alpha)
|
||||
plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", alpha=alpha)
|
||||
plt.axis(axes)
|
||||
plt.xlabel(r"$x_1$", fontsize=18)
|
||||
plt.ylabel(r"$x_2$", fontsize=18, rotation=0)
|
||||
plt.figure(figsize=(11,4))
|
||||
plt.subplot(121)
|
||||
plot_decision_boundary(tree_clf, X, y)
|
||||
plt.title("Decision Tree", fontsize=14)
|
||||
plt.subplot(122)
|
||||
plot_decision_boundary(bag_clf, X, y)
|
||||
plt.title("Decision Trees with Bagging", fontsize=14)
|
||||
save_fig("baggingtree")
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
!split
|
||||
===== Making your own Bootstrap: Changing the Level of the Decision Tree =====
|
||||
|
||||
Let us bring up our good old boostrap example from the linear regression lectures. We change the linerar regression algorithm with
|
||||
a decision tree wth different depths and perform a bootstrap aggregate (in this case we perform as many bootstraps as data points $n$).
|
||||
!bc pycod
|
||||
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.pipeline import make_pipeline
|
||||
from sklearn.utils import resample
|
||||
from sklearn.tree import DecisionTreeRegressor
|
||||
|
||||
n = 100
|
||||
n_boostraps = 100
|
||||
maxdepth = 8
|
||||
|
||||
# Make data set.
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
error = np.zeros(maxdepth)
|
||||
bias = np.zeros(maxdepth)
|
||||
variance = np.zeros(maxdepth)
|
||||
polydegree = np.zeros(maxdepth)
|
||||
X_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.2)
|
||||
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
scaler = StandardScaler()
|
||||
scaler.fit(X_train)
|
||||
X_train_scaled = scaler.transform(X_train)
|
||||
X_test_scaled = scaler.transform(X_test)
|
||||
|
||||
# we produce a simple tree first as benchmark
|
||||
simpletree = DecisionTreeRegressor(max_depth=3)
|
||||
simpletree.fit(X_train_scaled, y_train)
|
||||
simpleprediction = simpletree.predict(X_test_scaled)
|
||||
for degree in range(1,maxdepth):
|
||||
model = DecisionTreeRegressor(max_depth=degree)
|
||||
y_pred = np.empty((y_test.shape[0], n_boostraps))
|
||||
for i in range(n_boostraps):
|
||||
x_, y_ = resample(X_train_scaled, y_train)
|
||||
model.fit(x_, y_)
|
||||
y_pred[:, i] = model.predict(X_test_scaled)#.ravel()
|
||||
|
||||
polydegree[degree] = degree
|
||||
error[degree] = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )
|
||||
bias[degree] = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )
|
||||
variance[degree] = np.mean( np.var(y_pred, axis=1, keepdims=True) )
|
||||
print('Polynomial degree:', degree)
|
||||
print('Error:', error[degree])
|
||||
print('Bias^2:', bias[degree])
|
||||
print('Var:', variance[degree])
|
||||
print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree]))
|
||||
|
||||
mse_simpletree= np.mean( np.mean((y_test - simpleprediction)**2)
|
||||
print(mse_simpletree)
|
||||
plt.xlim(1,maxdepth)
|
||||
plt.plot(polydegree, error, label='MSE')
|
||||
plt.plot(polydegree, bias, label='bias')
|
||||
plt.plot(polydegree, variance, label='Variance')
|
||||
plt.legend()
|
||||
save_fig("baggingboot")
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
!split
|
||||
===== Why Voting? =====
|
||||
|
||||
@@ -1828,6 +1534,122 @@ for clf in (log_clf, rnd_clf, svm_clf, voting_clf):
|
||||
|
||||
|
||||
|
||||
!split
|
||||
===== Bagging =====
|
||||
|
||||
The _plain_ decision trees suffer from high
|
||||
variance. This means that if we split the training data into two parts
|
||||
at random, and fit a decision tree to both halves, the results that we
|
||||
get could be quite different. In contrast, a procedure with low
|
||||
variance will yield similar results if applied repeatedly to distinct
|
||||
data sets; linear regression tends to have low variance, if the ratio
|
||||
of $n$ to $p$ is moderately large.
|
||||
|
||||
_Bootstrap aggregation_, or just _bagging_, is a
|
||||
general-purpose procedure for reducing the variance of a statistical
|
||||
learning method.
|
||||
|
||||
|
||||
!split
|
||||
===== More bagging =====
|
||||
|
||||
Bagging typically results in improved accuracy
|
||||
over prediction using a single tree. Unfortunately, however, it can be
|
||||
difficult to interpret the resulting model. Recall that one of the
|
||||
advantages of decision trees is the attractive and easily interpreted
|
||||
diagram that results.
|
||||
|
||||
However, when we bag a large number of trees, it is no longer
|
||||
possible to represent the resulting statistical learning procedure
|
||||
using a single tree, and it is no longer clear which variables are
|
||||
most important to the procedure. Thus, bagging improves prediction
|
||||
accuracy at the expense of interpretability. Although the collection
|
||||
of bagged trees is much more difficult to interpret than a single
|
||||
tree, one can obtain an overall summary of the importance of each
|
||||
predictor using the MSE (for bagging regression trees) or the Gini
|
||||
index (for bagging classification trees). In the case of bagging
|
||||
regression trees, we can record the total amount that the MSE is
|
||||
decreased due to splits over a given predictor, averaged over all $B$ possible
|
||||
trees. A large value indicates an important predictor. Similarly, in
|
||||
the context of bagging classification trees, we can add up the total
|
||||
amount that the Gini index is decreased by splits over a given
|
||||
predictor, averaged over all $B$ trees.
|
||||
|
||||
|
||||
|
||||
!split
|
||||
===== Making your own Bootstrap: Changing the Level of the Decision Tree =====
|
||||
|
||||
Let us bring up our good old boostrap example from the linear regression lectures. We change the linerar regression algorithm with
|
||||
a decision tree wth different depths and perform a bootstrap aggregate (in this case we perform as many bootstraps as data points $n$).
|
||||
!bc pycod
|
||||
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.pipeline import make_pipeline
|
||||
from sklearn.utils import resample
|
||||
from sklearn.tree import DecisionTreeRegressor
|
||||
|
||||
n = 100
|
||||
n_boostraps = 100
|
||||
maxdepth = 8
|
||||
|
||||
# Make data set.
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
error = np.zeros(maxdepth)
|
||||
bias = np.zeros(maxdepth)
|
||||
variance = np.zeros(maxdepth)
|
||||
polydegree = np.zeros(maxdepth)
|
||||
X_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.2)
|
||||
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
scaler = StandardScaler()
|
||||
scaler.fit(X_train)
|
||||
X_train_scaled = scaler.transform(X_train)
|
||||
X_test_scaled = scaler.transform(X_test)
|
||||
|
||||
# we produce a simple tree first as benchmark
|
||||
simpletree = DecisionTreeRegressor(max_depth=3)
|
||||
simpletree.fit(X_train_scaled, y_train)
|
||||
simpleprediction = simpletree.predict(X_test_scaled)
|
||||
for degree in range(1,maxdepth):
|
||||
model = DecisionTreeRegressor(max_depth=degree)
|
||||
y_pred = np.empty((y_test.shape[0], n_boostraps))
|
||||
for i in range(n_boostraps):
|
||||
x_, y_ = resample(X_train_scaled, y_train)
|
||||
model.fit(x_, y_)
|
||||
y_pred[:, i] = model.predict(X_test_scaled)#.ravel()
|
||||
|
||||
polydegree[degree] = degree
|
||||
error[degree] = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )
|
||||
bias[degree] = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )
|
||||
variance[degree] = np.mean( np.var(y_pred, axis=1, keepdims=True) )
|
||||
print('Polynomial degree:', degree)
|
||||
print('Error:', error[degree])
|
||||
print('Bias^2:', bias[degree])
|
||||
print('Var:', variance[degree])
|
||||
print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree]))
|
||||
|
||||
mse_simpletree= np.mean( np.mean((y_test - simpleprediction)**2)
|
||||
print(mse_simpletree)
|
||||
plt.xlim(1,maxdepth)
|
||||
plt.plot(polydegree, error, label='MSE')
|
||||
plt.plot(polydegree, bias, label='bias')
|
||||
plt.plot(polydegree, variance, label='Variance')
|
||||
plt.legend()
|
||||
save_fig("baggingboot")
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
!split
|
||||
===== Random forests =====
|
||||
@@ -1902,19 +1724,14 @@ cancer = load_breast_cancer()
|
||||
X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
|
||||
print(X_train.shape)
|
||||
print(X_test.shape)
|
||||
#define methods
|
||||
# Logistic Regression
|
||||
logreg = LogisticRegression(solver='lbfgs')
|
||||
logreg.fit(X_train, y_train)
|
||||
print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
|
||||
# Support vector machine
|
||||
svm = SVC(gamma='auto', C=100)
|
||||
svm.fit(X_train, y_train)
|
||||
print("Test set accuracy with SVM: {:.2f}".format(svm.score(X_test,y_test)))
|
||||
# Decision Trees
|
||||
deep_tree_clf = DecisionTreeClassifier(max_depth=None)
|
||||
deep_tree_clf.fit(X_train, y_train)
|
||||
print("Test set accuracy with Decision Trees: {:.2f}".format(deep_tree_clf.score(X_test,y_test)))
|
||||
#now scale the data
|
||||
#Scale the data
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
scaler = StandardScaler()
|
||||
scaler.fit(X_train)
|
||||
|
||||
Reference in New Issue
Block a user