diff --git a/doc/HandWrittenNotes/2022/NotesNov102022.pdf b/doc/HandWrittenNotes/2022/NotesNov102022.pdf new file mode 100644 index 000000000..d2e6c2e05 Binary files /dev/null and b/doc/HandWrittenNotes/2022/NotesNov102022.pdf differ diff --git a/doc/pub/week45/html/._week45-bs000.html b/doc/pub/week45/html/._week45-bs000.html index b661511d2..2de25d156 100644 --- a/doc/pub/week45/html/._week45-bs000.html +++ b/doc/pub/week45/html/._week45-bs000.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -212,7 +266,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs001.html b/doc/pub/week45/html/._week45-bs001.html index c20f22063..264344ad4 100644 --- a/doc/pub/week45/html/._week45-bs001.html +++ b/doc/pub/week45/html/._week45-bs001.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -217,7 +271,7 @@ MathJax.Hub.Config({
  • 10
  • 11
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs002.html b/doc/pub/week45/html/._week45-bs002.html index 6c1b2240e..30e566f36 100644 --- a/doc/pub/week45/html/._week45-bs002.html +++ b/doc/pub/week45/html/._week45-bs002.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -298,7 +352,7 @@ plt.show()
  • 11
  • 12
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs003.html b/doc/pub/week45/html/._week45-bs003.html index 34860ce24..c170225c8 100644 --- a/doc/pub/week45/html/._week45-bs003.html +++ b/doc/pub/week45/html/._week45-bs003.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -201,7 +255,7 @@ them with a factor.
  • 12
  • 13
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs004.html b/doc/pub/week45/html/._week45-bs004.html index 525e357a6..8f8ca8705 100644 --- a/doc/pub/week45/html/._week45-bs004.html +++ b/doc/pub/week45/html/._week45-bs004.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -235,7 +289,7 @@ $$
  • 13
  • 14
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs005.html b/doc/pub/week45/html/._week45-bs005.html index 93e27fdc5..ed4a0708b 100644 --- a/doc/pub/week45/html/._week45-bs005.html +++ b/doc/pub/week45/html/._week45-bs005.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -208,7 +262,7 @@ at the internal nodes, and the predictions at the terminal nodes.
  • 14
  • 15
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs006.html b/doc/pub/week45/html/._week45-bs006.html index 1dc7094a6..7fa8432b5 100644 --- a/doc/pub/week45/html/._week45-bs006.html +++ b/doc/pub/week45/html/._week45-bs006.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -232,7 +286,7 @@ for \( \beta \) gives us an equation for \( \gamma \). This is a non-linear equa
  • 15
  • 16
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs007.html b/doc/pub/week45/html/._week45-bs007.html index 6ecf6e879..820ff8473 100644 --- a/doc/pub/week45/html/._week45-bs007.html +++ b/doc/pub/week45/html/._week45-bs007.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -222,7 +276,7 @@ $$
  • 16
  • 17
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs008.html b/doc/pub/week45/html/._week45-bs008.html index 78009d5cc..7b249a845 100644 --- a/doc/pub/week45/html/._week45-bs008.html +++ b/doc/pub/week45/html/._week45-bs008.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -216,7 +270,7 @@ $$
  • 17
  • 18
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs009.html b/doc/pub/week45/html/._week45-bs009.html index 03afa6a26..59de97a35 100644 --- a/doc/pub/week45/html/._week45-bs009.html +++ b/doc/pub/week45/html/._week45-bs009.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -233,7 +287,7 @@ $$
  • 18
  • 19
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs010.html b/doc/pub/week45/html/._week45-bs010.html index 485370ad7..20ec202f2 100644 --- a/doc/pub/week45/html/._week45-bs010.html +++ b/doc/pub/week45/html/._week45-bs010.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -212,7 +266,7 @@ $$
  • 19
  • 20
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs011.html b/doc/pub/week45/html/._week45-bs011.html index 42de926b5..2a65082f1 100644 --- a/doc/pub/week45/html/._week45-bs011.html +++ b/doc/pub/week45/html/._week45-bs011.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -226,7 +280,7 @@ observations that are missed in the previous iterations.
  • 20
  • 21
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs012.html b/doc/pub/week45/html/._week45-bs012.html index f3cfc221a..4a9c94890 100644 --- a/doc/pub/week45/html/._week45-bs012.html +++ b/doc/pub/week45/html/._week45-bs012.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -234,6 +288,8 @@ plt.show()
  • 20
  • 21
  • 22
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs013.html b/doc/pub/week45/html/._week45-bs013.html index d6c3e240a..7fd1bac56 100644 --- a/doc/pub/week45/html/._week45-bs013.html +++ b/doc/pub/week45/html/._week45-bs013.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -204,6 +258,9 @@ function was the least squares function.
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs014.html b/doc/pub/week45/html/._week45-bs014.html index 2d6004163..e573078d8 100644 --- a/doc/pub/week45/html/._week45-bs014.html +++ b/doc/pub/week45/html/._week45-bs014.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -221,6 +275,10 @@ $$
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs015.html b/doc/pub/week45/html/._week45-bs015.html index 4b3b33aeb..6279f2d9a 100644 --- a/doc/pub/week45/html/._week45-bs015.html +++ b/doc/pub/week45/html/._week45-bs015.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -204,6 +258,11 @@ $$
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs016.html b/doc/pub/week45/html/._week45-bs016.html index 53c864ee1..bee52a914 100644 --- a/doc/pub/week45/html/._week45-bs016.html +++ b/doc/pub/week45/html/._week45-bs016.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -212,6 +266,12 @@ $$
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • 26
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs017.html b/doc/pub/week45/html/._week45-bs017.html index 68a0e7d2d..1a322d0ca 100644 --- a/doc/pub/week45/html/._week45-bs017.html +++ b/doc/pub/week45/html/._week45-bs017.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -253,6 +307,13 @@ plt.show()
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • 26
  • +
  • 27
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs018.html b/doc/pub/week45/html/._week45-bs018.html index 2787cc51c..299e0b6f1 100644 --- a/doc/pub/week45/html/._week45-bs018.html +++ b/doc/pub/week45/html/._week45-bs018.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -251,6 +305,14 @@ plt.show()
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • 26
  • +
  • 27
  • +
  • 28
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs019.html b/doc/pub/week45/html/._week45-bs019.html index 0cde3b833..28dc926b6 100644 --- a/doc/pub/week45/html/._week45-bs019.html +++ b/doc/pub/week45/html/._week45-bs019.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -203,6 +257,15 @@ sketch for efficient proposal calculation. It introduces a novel sparsity-aware
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • 26
  • +
  • 27
  • +
  • 28
  • +
  • 29
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs020.html b/doc/pub/week45/html/._week45-bs020.html index 562c04cf5..0fa91e64e 100644 --- a/doc/pub/week45/html/._week45-bs020.html +++ b/doc/pub/week45/html/._week45-bs020.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -251,6 +305,16 @@ plt.show()
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • 26
  • +
  • 27
  • +
  • 28
  • +
  • 29
  • +
  • 30
  • +
  • ...
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs021.html b/doc/pub/week45/html/._week45-bs021.html index 12d57f114..940a5aff7 100644 --- a/doc/pub/week45/html/._week45-bs021.html +++ b/doc/pub/week45/html/._week45-bs021.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -261,6 +315,18 @@ plt.show()
  • 20
  • 21
  • 22
  • +
  • 23
  • +
  • 24
  • +
  • 25
  • +
  • 26
  • +
  • 27
  • +
  • 28
  • +
  • 29
  • +
  • 30
  • +
  • 31
  • +
  • ...
  • +
  • 40
  • +
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs022.html b/doc/pub/week45/html/._week45-bs022.html index f78ceea8b..7e5aac940 100644 --- a/doc/pub/week45/html/._week45-bs022.html +++ b/doc/pub/week45/html/._week45-bs022.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,29 +223,35 @@ MathJax.Hub.Config({

     

     

     

    -

    The Table

    +

    Support Vector Machines, overarching aims

    -
    -
    - - - - - - - - - - - - - - - - -
    Grade Trend Hours slept Hours Studied Grade
    Above Low High Above
    Below High Low Below
    Above Low High Above
    Above High High Above
    Below Low High Below
    Above Low Low Below
    Below High High Below
    Below Low High Below
    Above Low Low Below
    Above High High Above
    -
    -
    +

    A Support Vector Machine (SVM) is a very powerful and versatile +Machine Learning method, capable of performing linear or nonlinear +classification, regression, and even outlier detection. It is one of +the most popular models in Machine Learning, and anyone interested in +Machine Learning should have it in their toolbox. SVMs are +particularly well suited for classification of complex but small-sized or +medium-sized datasets. +

    + +

    The case with two well-separated classes only can be understood in an +intuitive way in terms of lines in a two-dimensional space separating +the two classes (see figure below). +

    + +

    The basic mathematics behind the SVM is however less familiar to most of us. +It relies on the definition of hyperplanes and the +definition of a margin which separates classes (in case of +classification problems) of variables. It is also used for regression +problems. +

    + +

    With SVMs we distinguish between hard margin and soft margins. The +latter introduces a so-called softening parameter to be discussed +below. We distinguish also between linear and non-linear +approaches. The latter are the most frequent ones since it is rather +unlikely that we can separate classes easily by say straight lines. +

    @@ -440,7 +278,7 @@ MathJax.Hub.Config({

  • 31
  • 32
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs023.html b/doc/pub/week45/html/._week45-bs023.html index 0017eee72..83cd00460 100644 --- a/doc/pub/week45/html/._week45-bs023.html +++ b/doc/pub/week45/html/._week45-bs023.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,17 +223,106 @@ MathJax.Hub.Config({

     

     

     

    -

    Computing the various Gini Indices

    +

    Hyperplanes and all that

    -

    In computations we will translate all classes into numbers. Being -these binary classes, they can easily be split into ones and zeros. +

    The theory behind support vector machines (SVM hereafter) is based on +the mathematical description of so-called hyperplanes. Let us start +with a two-dimensional case. This will also allow us to introduce our +first SVM examples. These will be tailored to the case of two specific +classes, as displayed in the figure here based on the usage of the petal data.

    -
    -
    - -

    See handwritten notes for Thursday November 11

    +

    We assume here that our data set can be well separated into two +domains, where a straight line does the job in the separating the two +classes. Here the two classes are represented by either squares or +circles. +

    + + +
    +
    +
    +
    +
    +
    from sklearn import datasets
    +from sklearn.svm import SVC, LinearSVC
    +from sklearn.linear_model import SGDClassifier
    +from sklearn.preprocessing import StandardScaler
    +import matplotlib
    +import matplotlib.pyplot as plt
    +plt.rcParams['axes.labelsize'] = 14
    +plt.rcParams['xtick.labelsize'] = 12
    +plt.rcParams['ytick.labelsize'] = 12
    +
    +
    +iris = datasets.load_iris()
    +X = iris["data"][:, (2, 3)]  # petal length, petal width
    +y = iris["target"]
    +
    +setosa_or_versicolor = (y == 0) | (y == 1)
    +X = X[setosa_or_versicolor]
    +y = y[setosa_or_versicolor]
    +
    +
    +
    +C = 5
    +alpha = 1 / (C * len(X))
    +
    +lin_clf = LinearSVC(loss="hinge", C=C, random_state=42)
    +svm_clf = SVC(kernel="linear", C=C)
    +sgd_clf = SGDClassifier(loss="hinge", learning_rate="constant", eta0=0.001, alpha=alpha,
    +                        max_iter=100000, random_state=42)
    +
    +scaler = StandardScaler()
    +X_scaled = scaler.fit_transform(X)
    +
    +lin_clf.fit(X_scaled, y)
    +svm_clf.fit(X_scaled, y)
    +sgd_clf.fit(X_scaled, y)
    +
    +print("LinearSVC:                   ", lin_clf.intercept_, lin_clf.coef_)
    +print("SVC:                         ", svm_clf.intercept_, svm_clf.coef_)
    +print("SGDClassifier(alpha={:.5f}):".format(sgd_clf.alpha), sgd_clf.intercept_, sgd_clf.coef_)
    +
    +# Compute the slope and bias of each decision boundary
    +w1 = -lin_clf.coef_[0, 0]/lin_clf.coef_[0, 1]
    +b1 = -lin_clf.intercept_[0]/lin_clf.coef_[0, 1]
    +w2 = -svm_clf.coef_[0, 0]/svm_clf.coef_[0, 1]
    +b2 = -svm_clf.intercept_[0]/svm_clf.coef_[0, 1]
    +w3 = -sgd_clf.coef_[0, 0]/sgd_clf.coef_[0, 1]
    +b3 = -sgd_clf.intercept_[0]/sgd_clf.coef_[0, 1]
    +
    +# Transform the decision boundary lines back to the original scale
    +line1 = scaler.inverse_transform([[-10, -10 * w1 + b1], [10, 10 * w1 + b1]])
    +line2 = scaler.inverse_transform([[-10, -10 * w2 + b2], [10, 10 * w2 + b2]])
    +line3 = scaler.inverse_transform([[-10, -10 * w3 + b3], [10, 10 * w3 + b3]])
    +
    +# Plot all three decision boundaries
    +plt.figure(figsize=(11, 4))
    +plt.plot(line1[:, 0], line1[:, 1], "k:", label="LinearSVC")
    +plt.plot(line2[:, 0], line2[:, 1], "b--", linewidth=2, label="SVC")
    +plt.plot(line3[:, 0], line3[:, 1], "r-", label="SGDClassifier")
    +plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs") # label="Iris-Versicolor"
    +plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo") # label="Iris-Setosa"
    +plt.xlabel("Petal length", fontsize=14)
    +plt.ylabel("Petal width", fontsize=14)
    +plt.legend(loc="upper center", fontsize=14)
    +plt.axis([0, 5.5, 0, 2])
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    @@ -430,7 +351,7 @@ these binary classes, they can easily be split into ones and zeros.
  • 32
  • 33
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs024.html b/doc/pub/week45/html/._week45-bs024.html index 77b337b16..0ce1060e5 100644 --- a/doc/pub/week45/html/._week45-bs024.html +++ b/doc/pub/week45/html/._week45-bs024.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,14 +223,32 @@ MathJax.Hub.Config({

     

     

     

    -

    Computing the various Gini Indices, Hours slept

    +

    What is a hyperplane?

    -
    -
    - -

    See handwritten notes for Thursday November 11

    -
    -
    +

    The aim of the SVM algorithm is to find a hyperplane in a +\( p \)-dimensional space, where \( p \) is the number of features that +distinctly classifies the data points. +

    + +

    In a \( p \)-dimensional space, a hyperplane is what we call an affine subspace of dimension of \( p-1 \). +As an example, in two dimension, a hyperplane is simply as straight line while in three dimensions it is +a two-dimensional subspace, or stated simply, a plane. +

    + +

    In two dimensions, with the variables \( x_1 \) and \( x_2 \), the hyperplane is defined as

    +$$ +b+w_1x_1+w_2x_2=0, +$$ + +

    where \( b \) is the intercept and \( w_1 \) and \( w_2 \) define the elements of a vector orthogonal to the line +\( b+w_1x_1+w_2x_2=0 \). +In two dimensions we define the vectors \( \boldsymbol{x} =[x1,x2] \) and \( \boldsymbol{w}=[w1,w2] \). +We can then rewrite the above equation as +

    + +$$ +\boldsymbol{x}^T\boldsymbol{w}+b=0. +$$

    @@ -426,7 +276,7 @@ MathJax.Hub.Config({

  • 33
  • 34
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs025.html b/doc/pub/week45/html/._week45-bs025.html index 23a2909ec..3a4e8341f 100644 --- a/doc/pub/week45/html/._week45-bs025.html +++ b/doc/pub/week45/html/._week45-bs025.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,17 +223,45 @@ MathJax.Hub.Config({

     

     

     

    -

    Computing the various Gini Indices, Hours studied

    +

    A \( p \)-dimensional space of features

    -
    -
    - -

    See handwritten notes for Thursday November 11

    -
    -
    +

    We limit ourselves to two classes of outputs \( y_i \) and assign these classes the values \( y_i = \pm 1 \). +In a \( p \)-dimensional space of say \( p \) features we have a hyperplane defines as +

    +$$ +b+wx_1+w_2x_2+\dots +w_px_p=0. +$$ +

    If we define a +matrix \( \boldsymbol{X}=\left[\boldsymbol{x}_1,\boldsymbol{x}_2,\dots, \boldsymbol{x}_p\right] \) +of dimension \( n\times p \), where \( n \) represents the observations for each feature and each vector \( x_i \) is a column vector of the matrix \( \boldsymbol{X} \), +

    +$$ +\boldsymbol{x}_i = \begin{bmatrix} x_{i1} \\ x_{i2} \\ \dots \\ \dots \\ x_{ip} \end{bmatrix}. +$$ -

    For final tree, see the above handwritten notes

    +

    If the above condition is not met for a given vector \( \boldsymbol{x}_i \) we have

    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} >0, +$$ + +

    if our output \( y_i=1 \). +In this case we say that \( \boldsymbol{x}_i \) lies on one of the sides of the hyperplane and if +

    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} < 0, +$$ + +

    for the class of observations \( y_i=-1 \), +then \( \boldsymbol{x}_i \) lies on the other side. +

    + +

    Equivalently, for the two classes of observations we have

    +$$ +y_i\left(b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip}\right) > 0. +$$ + +

    When we try to separate hyperplanes, if it exists, we can use it to construct a natural classifier: a test observation is assigned a given class depending on which side of the hyperplane it is located.

    @@ -428,7 +288,7 @@ MathJax.Hub.Config({

  • 34
  • 35
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs026.html b/doc/pub/week45/html/._week45-bs026.html index fa9dc631c..24b578f84 100644 --- a/doc/pub/week45/html/._week45-bs026.html +++ b/doc/pub/week45/html/._week45-bs026.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -390,98 +222,30 @@ MathJax.Hub.Config({

     

     

     

    - -

    A possible code using Scikit-Learn

    + +

    The two-dimensional case

    +

    Let us try to develop our intuition about SVMs by limiting ourselves to a two-dimensional +plane. To separate the two classes of data points, there are many +possible lines (hyperplanes if you prefer a more strict naming) +that could be chosen. Our objective is to find a +plane that has the maximum margin, i.e the maximum distance between +data points of both classes. Maximizing the margin distance provides +some reinforcement so that future data points can be classified with +more confidence. +

    - -
    -
    -
    -
    -
    -
    # Common imports
    -import numpy as np
    -import pandas as pd
    -import matplotlib.pyplot as plt
    -from sklearn.tree import DecisionTreeClassifier
    -from sklearn.model_selection import train_test_split
    -from sklearn.tree import export_graphviz
    -from sklearn.preprocessing import StandardScaler, OneHotEncoder
    -from sklearn.compose import ColumnTransformer
    -from IPython.display import Image 
    -from pydot import graph_from_dot_data
    -import os
    -
    -# Where to save the figures and data files
    -PROJECT_ROOT_DIR = "Results"
    -FIGURE_ID = "Results/FigureFiles"
    -DATA_ID = "DataFiles/"
    -
    -if not os.path.exists(PROJECT_ROOT_DIR):
    -    os.mkdir(PROJECT_ROOT_DIR)
    -
    -if not os.path.exists(FIGURE_ID):
    -    os.makedirs(FIGURE_ID)
    -
    -if not os.path.exists(DATA_ID):
    -    os.makedirs(DATA_ID)
    -
    -def image_path(fig_id):
    -    return os.path.join(FIGURE_ID, fig_id)
    -
    -def data_path(dat_id):
    -    return os.path.join(DATA_ID, dat_id)
    -
    -def save_fig(fig_id):
    -    plt.savefig(image_path(fig_id) + ".png", format='png')
    -
    -infile = open(data_path("grades.csv"),'r')
    -
    -# Read the experimental data with Pandas
    -from IPython.display import display
    -grades = pd.read_csv(infile,names = ('Trend','Sleep','Studied','Grade'))
    -grades = pd.DataFrame(grades)
    -
    -# Features and targets
    -X = grades.loc[:, grades.columns != 'Grade'].values
    -y = grades.loc[:, grades.columns == 'Grade'].values
    -
    -# Create the encoder.
    -encoder = OneHotEncoder(handle_unknown="ignore")
    -# Assume for simplicity all features are categorical.
    -encoder.fit(X)    
    -# Apply the encoder.
    -X = encoder.transform(X)
    -print(X)
    -# Then do a Classification tree
    -tree_clf = DecisionTreeClassifier(max_depth=2)
    -tree_clf.fit(X, y)
    -print("Train set accuracy with Decision Tree: {:.2f}".format(tree_clf.score(X,y)))
    -#transfer to a decision tree graph
    -export_graphviz(
    -    tree_clf,
    -    out_file="DataFiles/grade.dot",
    -    rounded=True,
    -    filled=True
    -)
    -cmd = 'dot -Tpng DataFiles/grade.dot -o DataFiles/grades.png'
    -os.system(cmd)
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    +

    What a linear classifier attempts to accomplish is to split the +feature space into two half spaces by placing a hyperplane between the +data points. This hyperplane will be our decision boundary. All +points on one side of the plane will belong to class one and all points +on the other side of the plane will belong to the second class two. +

    +

    Unfortunately there are many ways in which we can place a hyperplane +to divide the data. Below is an example of two candidate hyperplanes +for our data sample. +

    @@ -508,7 +272,7 @@ os.system(cmd)

  • 35
  • 36
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs027.html b/doc/pub/week45/html/._week45-bs027.html index fe7872523..1e93e6486 100644 --- a/doc/pub/week45/html/._week45-bs027.html +++ b/doc/pub/week45/html/._week45-bs027.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,61 +223,21 @@ MathJax.Hub.Config({

     

     

     

    -

    Visualizing Trees, More examples

    +

    Getting into the details

    - -
    -
    -
    -
    -
    -
    import os
    -from sklearn.datasets import load_breast_cancer
    -from sklearn.tree import DecisionTreeClassifier
    -from sklearn.model_selection import train_test_split
    -from sklearn.metrics import confusion_matrix
    -from sklearn.tree import export_graphviz
    +

    Let us define the function

    +$$ +f(x) = \boldsymbol{w}^T\boldsymbol{x}+b = 0, +$$ -from IPython.display import Image -from pydot import graph_from_dot_data -import pandas as pd -import numpy as np +

    as the function that determines the line \( L \) that separates two classes (our two features), see the figure here.

    +

    Any point defined by \( \boldsymbol{x}_i \) and \( \boldsymbol{x}_2 \) on the line \( L \) will satisfy \( \boldsymbol{w}^T(\boldsymbol{x}_1-\boldsymbol{x}_2)=0 \).

    -cancer = load_breast_cancer() -X = pd.DataFrame(cancer.data, columns=cancer.feature_names) -print(X) -y = pd.Categorical.from_codes(cancer.target, cancer.target_names) -y = pd.get_dummies(y) -print(y) -X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=1) -tree_clf = DecisionTreeClassifier(max_depth=5) -tree_clf.fit(X_train, y_train) - -export_graphviz( - tree_clf, - out_file="DataFiles/cancer.dot", - feature_names=cancer.feature_names, - class_names=cancer.target_names, - rounded=True, - filled=True -) -cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png' -os.system(cmd) -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    +

    The signed distance \( \delta \) from any point defined by a vector \( \boldsymbol{x} \) and a point \( \boldsymbol{x}_0 \) on the line \( L \) is then

    +$$ +\delta = \frac{1}{\vert\vert \boldsymbol{w}\vert\vert}(\boldsymbol{w}^T\boldsymbol{x}+b). +$$

    @@ -473,7 +265,7 @@ os.system(cmd)

  • 36
  • 37
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs028.html b/doc/pub/week45/html/._week45-bs028.html index 30bf163d3..443313424 100644 --- a/doc/pub/week45/html/._week45-bs028.html +++ b/doc/pub/week45/html/._week45-bs028.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,52 +223,26 @@ MathJax.Hub.Config({

     

     

     

    -

    Visualizing the Tree, The Moons

    +

    First attempt at a minimization approach

    - -
    -
    -
    -
    -
    -
    # Common imports
    -import numpy as np
    -from sklearn.model_selection import  train_test_split 
    -from sklearn.tree import DecisionTreeClassifier
    -from sklearn.datasets import make_moons
    -from sklearn.tree import export_graphviz
    -from pydot import graph_from_dot_data
    -import pandas as pd
    -import os
    +

    How do we find the parameter \( b \) and the vector \( \boldsymbol{w} \)? What we could +do is to define a cost function which now contains the set of all +misclassified points \( M \) and attempt to minimize this function +

    -np.random.seed(42) -X, y = make_moons(n_samples=100, noise=0.25, random_state=53) -X_train, X_test, y_train, y_test = train_test_split(X,y,random_state=0) -tree_clf = DecisionTreeClassifier(max_depth=5) -tree_clf.fit(X_train, y_train) +$$ +C(\boldsymbol{w},b) = -\sum_{i\in M} y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ -export_graphviz( - tree_clf, - out_file="DataFiles/moons.dot", - rounded=True, - filled=True -) -cmd = 'dot -Tpng DataFiles/moons.dot -o DataFiles/moons.png' -os.system(cmd) -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    +

    We could now for example define all values \( y_i =1 \) as misclassified in case we have \( \boldsymbol{w}^T\boldsymbol{x}_i+b < 0 \) and the opposite if we have \( y_i=-1 \). Taking the derivatives gives us

    +$$ +\frac{\partial C}{\partial b} = -\sum_{i\in M} y_i, +$$ + +

    and

    +$$ +\frac{\partial C}{\partial \boldsymbol{w}} = -\sum_{i\in M} y_ix_i. +$$

    @@ -464,7 +270,7 @@ os.system(cmd)

  • 37
  • 38
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs029.html b/doc/pub/week45/html/._week45-bs029.html index 05c540319..3a6264fed 100644 --- a/doc/pub/week45/html/._week45-bs029.html +++ b/doc/pub/week45/html/._week45-bs029.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,39 +223,19 @@ MathJax.Hub.Config({

     

     

     

    -

    Other ways of visualizing the trees

    +

    Solving the equations

    -

    Scikit-Learn has also another way to visualize the trees which is very useful, here with the Iris data.

    +

    We can now use the Newton-Raphson method or different variants of the gradient descent family (from plain gradient descent to various stochastic gradient descent approaches) to solve the equations

    +$$ +b \leftarrow b +\eta \frac{\partial C}{\partial b}, +$$ +

    and

    +$$ +\boldsymbol{w} \leftarrow \boldsymbol{w} +\eta \frac{\partial C}{\partial \boldsymbol{w}}, +$$ - -
    -
    -
    -
    -
    -
    from sklearn.datasets import load_iris
    -from sklearn import tree
    -X, y = load_iris(return_X_y=True)
    -tree_clf = tree.DecisionTreeClassifier()
    -tree_clf = tree_clf.fit(X, y)
    -# and then plot the tree
    -tree.plot_tree(tree_clf) 
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    - +

    where \( \eta \) is our by now well-known learning rate.

    @@ -450,7 +262,7 @@ tree.plot_tree(tree_clf)

  • 38
  • 39
  • ...
  • -
  • 79
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs030.html b/doc/pub/week45/html/._week45-bs030.html index c4b42d0a1..eb5812818 100644 --- a/doc/pub/week45/html/._week45-bs030.html +++ b/doc/pub/week45/html/._week45-bs030.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,27 +223,20 @@ MathJax.Hub.Config({

     

     

     

    -

    Printing out as text

    +

    Code Example

    -

    Alternatively, the tree can also be exported in textual format with the function exporttext. -This method doesn’t require the installation of external libraries and is more compact: +

    The equations we discussed above can be coded rather easily (the +framework is similar to what we developed for logistic +regression). We are going to set up a simple case with two classes only and we want to find a line which separates them the best possible way.

    -
    -
    from sklearn.datasets import load_iris
    -from sklearn.tree import DecisionTreeClassifier
    -from sklearn.tree import export_text
    -iris = load_iris()
    -decision_tree = DecisionTreeClassifier(random_state=0, max_depth=2)
    -decision_tree = decision_tree.fit(iris.data, iris.target)
    -r = export_text(decision_tree, feature_names=iris['feature_names'])
    -print(r)
    +  
     
    @@ -452,8 +277,6 @@ r = export_text(decision_tree, feature_names
  • 38
  • 39
  • 40
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs031.html b/doc/pub/week45/html/._week45-bs031.html index fac19847c..8997c7970 100644 --- a/doc/pub/week45/html/._week45-bs031.html +++ b/doc/pub/week45/html/._week45-bs031.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,17 +223,17 @@ MathJax.Hub.Config({

     

     

     

    -

    Algorithms for Setting up Decision Trees

    +

    Problems with the Simpler Approach

    -

    Two algorithms stand out in the set up of decision trees:

    -
      -
    1. The CART (Classification And Regression Tree) algorithm for both classification and regression
    2. -
    3. The ID3 algorithm based on the computation of the information gain for classification
    4. -
    -

    We discuss both algorithms with applications here. The popular library -Scikit-Learn uses the CART algorithm. For classification problems -you can use either the gini index or the entropy to split a tree -in two branches. +

    There are however problems with this approach, although it looks +pretty straightforward to implement. When running the above code, we see that we can easily end up with many diffeent lines which separate the two classes. +

    + +

    For small +gaps between the entries, we may also end up needing many iterations +before the solutions converge and if the data cannot be separated +properly into two distinct classes, we may not experience a converge +at all.

    @@ -427,9 +259,6 @@ in two branches.

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs032.html b/doc/pub/week45/html/._week45-bs032.html index 97c060217..4d36e3d67 100644 --- a/doc/pub/week45/html/._week45-bs032.html +++ b/doc/pub/week45/html/._week45-bs032.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,29 +223,43 @@ MathJax.Hub.Config({

     

     

     

    -

    The CART algorithm for Classification

    +

    A better approach

    -

    For classification, the CART algorithm splits the data set in two subsets using a single feature \( k \) and a threshold \( t_k \). -This could be for example a threshold set by a number below a certain circumference of a malign tumor. +

    A better approach is rather to try to define a large margin between +the two classes (if they are well separated from the beginning).

    -

    How do we find these two quantities? -We search for the pair \( (k,t_k) \) that produces the purest subset using for example the gini factor \( G \). -The cost function it tries to minimize is then +

    Thus, we wish to find a margin \( M \) with \( \boldsymbol{w} \) normalized to +\( \vert\vert \boldsymbol{w}\vert\vert =1 \) subject to the condition

    + $$ -C(k,t_k) = \frac{m_{\mathrm{left}}}{m}G_{\mathrm{left}}+ \frac{m_{\mathrm{right}}}{m}G_{\mathrm{right}}, +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, p. $$ -

    where \( G_{\mathrm{left/right}} \) measures the impurity of the left/right subset and \( m_{\mathrm{left/right}} \) - is the number of instances in the left/right subset -

    +

    All points are thus at a signed distance from the decision boundary defined by the line \( L \). The parameters \( b \) and \( w_1 \) and \( w_2 \) define this line.

    -

    Once it has successfully split the training set in two, it splits the subsets using the same logic, then the subsubsets -and so on, recursively. It stops recursing once it reaches the maximum depth (defined by the -\( max\_depth \) hyperparameter), or if it cannot find a split that will reduce impurity. A few other -hyperparameters control additional stopping conditions such as the \( min\_samples\_split \), -\( min\_samples\_leaf \), \( min\_weight\_fraction\_leaf \), and \( max\_leaf\_nodes \). +

    We seek thus the largest value \( M \) defined by

    +$$ +\frac{1}{\vert \vert \boldsymbol{w}\vert\vert}y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, n, +$$ + +

    or just

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M\vert \vert \boldsymbol{w}\vert\vert \hspace{0.1cm}\forall i. +$$ + +

    If we scale the equation so that \( \vert \vert \boldsymbol{w}\vert\vert = 1/M \), we have to find the minimum of +\( \boldsymbol{w}^T\boldsymbol{w}=\vert \vert \boldsymbol{w}\vert\vert \) (the norm) subject to the condition +

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq 1 \hspace{0.1cm}\forall i. +$$ + +

    We have thus defined our margin as the invers of the norm of +\( \boldsymbol{w} \). We want to minimize the norm in order to have a as large as +possible margin \( M \). Before we proceed, we need to remind ourselves +about Lagrangian multipliers.

    @@ -438,10 +284,6 @@ hyperparameters control additional stopping conditions such as the \( min\_sampl

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs033.html b/doc/pub/week45/html/._week45-bs033.html index 1ce0995b5..05582a354 100644 --- a/doc/pub/week45/html/._week45-bs033.html +++ b/doc/pub/week45/html/._week45-bs033.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,29 +223,53 @@ MathJax.Hub.Config({

     

     

     

    -

    The CART algorithm for Regression

    +

    A quick Reminder on Lagrangian Multipliers

    -

    The CART algorithm for regression works is similar to the one for classification except that instead of trying to split the -training set in a way that minimizes say the gini or entropy impurity, it now tries to split the training set in a way that minimizes our well-known mean-squared error (MSE). The cost function is now +

    Consider a function of three independent variables \( f(x,y,z) \) . For the function \( f \) to be an +extreme we have

    $$ -C(k,t_k) = \frac{m_{\mathrm{left}}}{m}\mathrm{MSE}_{\mathrm{left}}+ \frac{m_{\mathrm{right}}}{m}\mathrm{MSE}_{\mathrm{right}}. +df=0. $$ -

    Here the MSE for a specific node is defined as

    +

    A necessary and sufficient condition is

    $$ -\mathrm{MSE}_{\mathrm{node}}=\frac{1}{m_\mathrm{node}}\sum_{i\in \mathrm{node}}(\overline{y}_{\mathrm{node}}-y_i)^2, +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, $$ -

    with

    +

    due to

    $$ -\overline{y}_{\mathrm{node}}=\frac{1}{m_\mathrm{node}}\sum_{i\in \mathrm{node}}y_i, +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz. $$ -

    the mean value of all observations in a specific node.

    +

    In many problems the variables \( x,y,z \) are often subject to constraints (such as those above for the margin) +so that they are no longer all independent. It is possible at least in principle to use each +constraint to eliminate one variable +and to proceed with a new and smaller set of independent varables. +

    -

    Without any regularization, the regression task for decision trees, -just like for classification tasks, is prone to overfitting. +

    The use of so-called Lagrangian multipliers is an alternative technique when the elimination +of variables is incovenient or undesirable. Assume that we have an equation of constraint on +the variables \( x,y,z \) +

    +$$ +\phi(x,y,z) = 0, +$$ + +

    resulting in

    +$$ +d\phi = \frac{\partial \phi}{\partial x}dx+\frac{\partial \phi}{\partial y}dy+\frac{\partial \phi}{\partial z}dz =0. +$$ + +

    Now we cannot set anymore

    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ + +

    if \( df=0 \) is wanted +because there are now only two independent variables! Assume \( x \) and \( y \) are the independent +variables. +Then \( dz \) is no longer arbitrary.

    @@ -437,11 +293,6 @@ just like for classification tasks, is prone to overfitting.

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • 43
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs034.html b/doc/pub/week45/html/._week45-bs034.html index b9ca0528d..b9e993b1e 100644 --- a/doc/pub/week45/html/._week45-bs034.html +++ b/doc/pub/week45/html/._week45-bs034.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,46 +223,45 @@ MathJax.Hub.Config({

     

     

     

    -

    Computing the Gini index

    +

    Adding the Multiplier

    -

    The example we will look at is a classical one in many Machine -Learning applications. Based on various meteorological features, we -have several so-called attributes which decide whether we at the end -will do some outdoor activity like skiing, going for a bike ride etc -etc. The table here contains the feautures outlook, temperature, -humidity and wind. The target or output is whether we ride -(True=1) or whether we do something else that day (False=0). The -attributes for each feature are then sunny, overcast and rain for the -outlook, hot, cold and mild for temperature, high and normal for -humidity and weak and strong for wind. +

    However, we can add to

    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz, +$$ + +

    a multiplum of \( d\phi \), viz. \( \lambda d\phi \), resulting in

    +$$ +df+\lambda d\phi = (\frac{\partial f}{\partial z}+\lambda +\frac{\partial \phi}{\partial x})dx+(\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y})dy+ +(\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z})dz =0. +$$ + +

    Our multiplier is chosen so that

    +$$ +\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z} =0. +$$ + +

    We need to remember that we took \( dx \) and \( dy \) to be arbitrary and thus we must have

    +$$ +\frac{\partial f}{\partial x}+\lambda\frac{\partial \phi}{\partial x} =0, +$$ + +

    and

    +$$ +\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y} =0. +$$ + +

    When all these equations are satisfied, \( df=0 \). We have four unknowns, \( x,y,z \) and +\( \lambda \). Actually we want only \( x,y,z \), \( \lambda \) needs not to be determined, +it is therefore often called +Lagrange's undetermined multiplier. +If we have a set of constraints \( \phi_k \) we have the equations

    +$$ +\frac{\partial f}{\partial x_i}+\sum_k\lambda_k\frac{\partial \phi_k}{\partial x_i} =0. +$$ -

    The table here summarizes the various attributes and

    -
    -
    - - - - - - - - - - - - - - - - - - - - -
    Day Outlook Temperature Humidity Wind Ride
    1 Sunny Hot High Weak 0
    2 Sunny Hot High Strong 1
    3 Overcast Hot High Weak 1
    4 Rain Mild High Weak 1
    5 Rain Cool Normal Weak 1
    6 Rain Cool Normal Strong 0
    7 Overcast Cool Normal Strong 1
    8 Sunny Mild High Weak 0
    9 Sunny Cool Normal Weak 1
    10 Rain Mild Normal Weak 1
    11 Sunny Mild Normal Strong 1
    12 Overcast Mild High Strong 1
    13 Overcast Hot Normal Weak 1
    14 Rain Mild High Strong 0
    -
    -

    @@ -452,12 +283,6 @@ humidity and weak and strong for wind.

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • 43
  • -
  • 44
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs035.html b/doc/pub/week45/html/._week45-bs035.html index d834fc2e5..a035686f2 100644 --- a/doc/pub/week45/html/._week45-bs035.html +++ b/doc/pub/week45/html/._week45-bs035.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,97 +223,41 @@ MathJax.Hub.Config({

     

     

     

    -

    Simple Python Code to read in Data and perform Classification

    +

    Setting up the Problem

    +

    In order to solve the above problem, we define the following Lagrangian function to be minimized

    +$$ +{\cal L}(\lambda,b,\boldsymbol{w})=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-1\right], +$$ +

    where \( \lambda_i \) is a so-called Lagrange multiplier subject to the condition \( \lambda_i \geq 0 \).

    - -
    -
    -
    -
    -
    -
    # Common imports
    -import numpy as np
    -import pandas as pd
    -import matplotlib.pyplot as plt
    -from sklearn.tree import DecisionTreeClassifier
    -from sklearn.model_selection import train_test_split
    -from sklearn.tree import export_graphviz
    -from sklearn.preprocessing import StandardScaler, OneHotEncoder
    -from sklearn.compose import ColumnTransformer
    -from IPython.display import Image 
    -from pydot import graph_from_dot_data
    -import os
    +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ -# Where to save the figures and data files -PROJECT_ROOT_DIR = "Results" -FIGURE_ID = "Results/FigureFiles" -DATA_ID = "DataFiles/" +

    and

    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ -if not os.path.exists(PROJECT_ROOT_DIR): - os.mkdir(PROJECT_ROOT_DIR) +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ -if not os.path.exists(FIGURE_ID): - os.makedirs(FIGURE_ID) - -if not os.path.exists(DATA_ID): - os.makedirs(DATA_ID) - -def image_path(fig_id): - return os.path.join(FIGURE_ID, fig_id) - -def data_path(dat_id): - return os.path.join(DATA_ID, dat_id) - -def save_fig(fig_id): - plt.savefig(image_path(fig_id) + ".png", format='png') - -infile = open(data_path("rideclass.csv"),'r') - -# Read the experimental data with Pandas -from IPython.display import display -ridedata = pd.read_csv(infile,names = ('Outlook','Temperature','Humidity','Wind','Ride')) -ridedata = pd.DataFrame(ridedata) - -# Features and targets -X = ridedata.loc[:, ridedata.columns != 'Ride'].values -y = ridedata.loc[:, ridedata.columns == 'Ride'].values - -# Create the encoder. -encoder = OneHotEncoder(handle_unknown="ignore") -# Assume for simplicity all features are categorical. -encoder.fit(X) -# Apply the encoder. -X = encoder.transform(X) -print(X) -# Then do a Classification tree -tree_clf = DecisionTreeClassifier(max_depth=2) -tree_clf.fit(X, y) -print("Train set accuracy with Decision Tree: {:.2f}".format(tree_clf.score(X,y))) -#transfer to a decision tree graph -export_graphviz( - tree_clf, - out_file="DataFiles/ride.dot", - rounded=True, - filled=True -) -cmd = 'dot -Tpng DataFiles/cancer.dot -o DataFiles/cancer.png' -os.system(cmd) -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    +

    subject to the constraints \( \lambda_i\geq 0 \) and \( \sum_i\lambda_iy_i=0 \). +We must in addition satisfy the Karush-Kuhn-Tucker (KKT) condition +

    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -1\right] \hspace{0.1cm}\forall i. +$$ +
      +
    1. If \( \lambda_i > 0 \), then \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) and we say that \( x_i \) is on the boundary.
    2. +
    3. If \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)> 1 \), we say \( x_i \) is not on the boundary and we set \( \lambda_i=0 \).
    4. +
    +

    When \( \lambda_i > 0 \), the vectors \( \boldsymbol{x}_i \) are called support vectors. They are the vectors closest to the line (or hyperplane) and define the margin \( M \).

    @@ -502,13 +278,6 @@ os.system(cmd)

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • 43
  • -
  • 44
  • -
  • 45
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs036.html b/doc/pub/week45/html/._week45-bs036.html index d704ceffe..ebe46a37a 100644 --- a/doc/pub/week45/html/._week45-bs036.html +++ b/doc/pub/week45/html/._week45-bs036.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,99 +223,27 @@ MathJax.Hub.Config({

     

     

     

    -

    Computing the Gini Factor

    +

    The problem to solve

    -

    The above functions (gini, entropy and misclassification error) are -important components of the so-called CART algorithm. We will discuss -this algorithm below after we have discussed the information gain -algorithm ID3. +

    We can rewrite

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    and its constraints in terms of a matrix-vector problem where we minimize w.r.t. \( \lambda \) the following problem

    +$$ +\frac{1}{2} \boldsymbol{\lambda}^T\begin{bmatrix} y_1y_1\boldsymbol{x}_1^T\boldsymbol{x}_1 & y_1y_2\boldsymbol{x}_1^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_1^T\boldsymbol{x}_n \\ +y_2y_1\boldsymbol{x}_2^T\boldsymbol{x}_1 & y_2y_2\boldsymbol{x}_2^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_2^T\boldsymbol{x}_n \\ +\dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots \\ +y_ny_1\boldsymbol{x}_n^T\boldsymbol{x}_1 & y_ny_2\boldsymbol{x}_n^T\boldsymbol{x}_2 & \dots & \dots & y_ny_n\boldsymbol{x}_n^T\boldsymbol{x}_n \\ +\end{bmatrix}\boldsymbol{\lambda}-\mathbb{1}\boldsymbol{\lambda}, +$$ + +

    subject to \( \boldsymbol{y}^T\boldsymbol{\lambda}=0 \). Here we defined the vectors \( \boldsymbol{\lambda} =[\lambda_1,\lambda_2,\dots,\lambda_n] \) and +\( \boldsymbol{y}=[y_1,y_2,\dots,y_n] \).

    -

    In the example here we have converted all our attributes into numerical values \( 0,1,2 \) etc.

    - - - -
    -
    -
    -
    -
    -
    # Split a dataset based on an attribute and an attribute value
    -def test_split(index, value, dataset):
    -	left, right = list(), list()
    -	for row in dataset:
    -		if row[index] < value:
    -			left.append(row)
    -		else:
    -			right.append(row)
    -	return left, right
    - 
    -# Calculate the Gini index for a split dataset
    -def gini_index(groups, classes):
    -	# count all samples at split point
    -	n_instances = float(sum([len(group) for group in groups]))
    -	# sum weighted Gini index for each group
    -	gini = 0.0
    -	for group in groups:
    -		size = float(len(group))
    -		# avoid divide by zero
    -		if size == 0:
    -			continue
    -		score = 0.0
    -		# score the group based on the score for each class
    -		for class_val in classes:
    -			p = [row[-1] for row in group].count(class_val) / size
    -			score += p * p
    -		# weight the group score by its relative size
    -		gini += (1.0 - score) * (size / n_instances)
    -	return gini
    -
    -# Select the best split point for a dataset
    -def get_split(dataset):
    -	class_values = list(set(row[-1] for row in dataset))
    -	b_index, b_value, b_score, b_groups = 999, 999, 999, None
    -	for index in range(len(dataset[0])-1):
    -		for row in dataset:
    -			groups = test_split(index, row[index], dataset)
    -			gini = gini_index(groups, class_values)
    -			print('X%d < %.3f Gini=%.3f' % ((index+1), row[index], gini))
    -			if gini < b_score:
    -				b_index, b_value, b_score, b_groups = index, row[index], gini, groups
    -	return {'index':b_index, 'value':b_value, 'groups':b_groups}
    - 
    -dataset = [[0,0,0,0,0],
    -            [0,0,0,1,1],
    -            [1,0,0,0,1],
    -            [2,1,0,0,1],
    -            [2,2,1,0,1],
    -            [2,2,1,1,0],
    -            [1,2,1,1,1],
    -            [0,1,0,0,0],
    -            [0,2,1,0,1],
    -            [2,1,1,0,1],
    -            [0,1,1,1,1],
    -            [1,1,0,1,1],
    -            [1,0,1,0,1],
    -            [2,1,0,1,0]]
    -
    -split = get_split(dataset)
    -print('Split: [X%d < %.3f]' % ((split['index']+1), split['value']))
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    - -

    diff --git a/doc/pub/week45/html/._week45-bs037.html b/doc/pub/week45/html/._week45-bs037.html index 8e9de3ab8..86a82c657 100644 --- a/doc/pub/week45/html/._week45-bs037.html +++ b/doc/pub/week45/html/._week45-bs037.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,37 +223,36 @@ MathJax.Hub.Config({

     

     

     

    -

    Entropy and the ID3 algorithm

    +

    The last steps

    -

    The ID3 algorithm learns decision trees by constructing -them in a top down way, beginning with the question which attribute should be tested at the root of the tree? +

    Solving the above problem, yields the values of \( \lambda_i \). +To find the coefficients of your hyperplane we need simply to compute

    +$$ +\boldsymbol{w}=\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ -
      -
    1. Each instance attribute is evaluated using a statistical test to determine how well it alone classifies the training examples.
    2. -
    3. The best attribute is selected and used as the test at the root node of the tree.
    4. -
    5. A descendant of the root node is then created for each possible value of this attribute.
    6. -
    7. Training examples are sorted to the appropriate descendant node.
    8. -
    9. The entire process is then repeated using the training examples associated with each descendant node to select the best attribute to test at that point in the tree.
    10. -
    11. This forms a greedy search for an acceptable decision tree, in which the algorithm never backtracks to reconsider earlier choices.
    12. -
    -

    The ID3 algorithm selects which attribute to test at each node in the -tree. -

    +

    With our vector \( \boldsymbol{w} \) we can in turn find the value of the intercept \( b \) (here in two dimensions) via

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ -

    We would like to select the attribute that is most useful for classifying -examples. -

    +

    resulting in

    +$$ +b = \frac{1}{y_i}-\boldsymbol{w}^T\boldsymbol{x}_i, +$$ -

    What is a good quantitative measure of the worth of an attribute?

    +

    or if we write it out in terms of the support vectors only, with \( N_s \) being their number, we have

    +$$ +b = \frac{1}{N_s}\sum_{j\in N_s}\left(y_j-\sum_{i=1}^n\lambda_iy_i\boldsymbol{x}_i^T\boldsymbol{x}_j\right). +$$ -

    Information gain measures how well a given attribute separates the -training examples according to their target classification. -

    +

    With our hyperplane coefficients we can use our classifier to assign any observation by simply using

    +$$ +y_i = \mathrm{sign}(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ -

    The ID3 algorithm uses this information gain measure to select among the candidate -attributes at each step while growing the tree. -

    +

    Below we discuss how to find the optimal values of \( \lambda_i \). Before we proceed however, we discuss now the so-called soft classifier.

    @@ -440,15 +271,6 @@ attributes at each step while growing the tree.

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • 43
  • -
  • 44
  • -
  • 45
  • -
  • 46
  • -
  • 47
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs038.html b/doc/pub/week45/html/._week45-bs038.html index 43f7a5b2d..a8691edf6 100644 --- a/doc/pub/week45/html/._week45-bs038.html +++ b/doc/pub/week45/html/._week45-bs038.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,70 +223,37 @@ MathJax.Hub.Config({

     

     

     

    -

    Cancer Data again now with Decision Trees and other Methods

    +

    A soft classifier

    - -
    -
    -
    -
    -
    -
    import matplotlib.pyplot as plt
    -import numpy as np
    -from sklearn.model_selection import  train_test_split 
    -from sklearn.datasets import load_breast_cancer
    -from sklearn.svm import SVC
    -from sklearn.linear_model import LogisticRegression
    -from sklearn.tree import DecisionTreeClassifier
    +

    Till now, the margin is strictly defined by the support vectors. This defines what is called a hard classifier, that is the margins are well defined.

    -# Load the data -cancer = load_breast_cancer() +

    Suppose now that classes overlap in feature space, as shown in the +figure here. One way to deal with this problem before we define the +so-called kernel approach, is to allow a kind of slack in the sense +that we allow some points to be on the wrong side of the margin. +

    -X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) -print(X_train.shape) -print(X_test.shape) -# Logistic Regression -logreg = LogisticRegression(solver='lbfgs') -logreg.fit(X_train, y_train) -print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) -# Support vector machine -svm = SVC(gamma='auto', C=100) -svm.fit(X_train, y_train) -print("Test set accuracy with SVM: {:.2f}".format(svm.score(X_test,y_test))) -# Decision Trees -deep_tree_clf = DecisionTreeClassifier(max_depth=None) -deep_tree_clf.fit(X_train, y_train) -print("Test set accuracy with Decision Trees: {:.2f}".format(deep_tree_clf.score(X_test,y_test))) -#now scale the data -from sklearn.preprocessing import StandardScaler -scaler = StandardScaler() -scaler.fit(X_train) -X_train_scaled = scaler.transform(X_train) -X_test_scaled = scaler.transform(X_test) -# Logistic Regression -logreg.fit(X_train_scaled, y_train) -print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test))) -# Support Vector Machine -svm.fit(X_train_scaled, y_train) -print("Test set accuracy SVM with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test))) -# Decision Trees -deep_tree_clf.fit(X_train_scaled, y_train) -print("Test set accuracy with Decision Trees and scaled data: {:.2f}".format(deep_tree_clf.score(X_test_scaled,y_test))) -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    +

    We introduce thus the so-called slack variables \( \boldsymbol{\xi} =[\xi_1,x_2,\dots,x_n] \) and +modify our previous equation +

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ +

    to

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i, +$$ + +

    with the requirement \( \xi_i\geq 0 \). The total violation is now \( \sum_i\xi \). +The value \( \xi_i \) in the constraint the last constraint corresponds to the amount by which the prediction +\( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) is on the wrong side of its margin. Hence by bounding the sum \( \sum_i \xi_i \), +we bound the total amount by which predictions fall on the wrong side of their margins. +

    + +

    Misclassifications occur when \( \xi_i > 1 \). Thus bounding the total sum by some value \( C \) bounds in turn the total number of +misclassifications. +

    @@ -472,16 +271,6 @@ deep_tree_clf.fit(X_train_scaled, y_train)

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • 43
  • -
  • 44
  • -
  • 45
  • -
  • 46
  • -
  • 47
  • -
  • 48
  • -
  • ...
  • -
  • 79
  • »
  • diff --git a/doc/pub/week45/html/._week45-bs039.html b/doc/pub/week45/html/._week45-bs039.html index 438a5bf3d..adde67a26 100644 --- a/doc/pub/week45/html/._week45-bs039.html +++ b/doc/pub/week45/html/._week45-bs039.html @@ -37,175 +37,10 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d @@ -303,83 +174,44 @@ MathJax.Hub.Config({ Contents @@ -391,93 +223,55 @@ MathJax.Hub.Config({

     

     

     

    -

    Another example, the moons again

    +

    Soft optmization problem

    - -
    -
    -
    -
    -
    -
    from __future__ import division, print_function, unicode_literals
    +

    This has in turn the consequences that we change our optmization problem to finding the minimum of

    +$$ +{\cal L}=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-(1-\xi_)\right]+C\sum_{i=1}^n\xi_i-\sum_{i=1}^n\gamma_i\xi_i, +$$ -# Common imports -import numpy as np -import os +

    subject to

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i \hspace{0.1cm}\forall i, +$$ -# to make this notebook's output stable across runs -np.random.seed(42) +

    with the requirement \( \xi_i\geq 0 \).

    -# To plot pretty figures -import matplotlib -import matplotlib.pyplot as plt -from matplotlib.colors import ListedColormap -plt.rcParams['axes.labelsize'] = 14 -plt.rcParams['xtick.labelsize'] = 12 -plt.rcParams['ytick.labelsize'] = 12 +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ +

    and

    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i, +$$ -from sklearn.svm import SVC -from sklearn import datasets -from sklearn.tree import DecisionTreeClassifier -from sklearn.datasets import make_moons -from sklearn.tree import export_graphviz +

    and

    +$$ +\lambda_i = C-\gamma_i \hspace{0.1cm}\forall i. +$$ -Xm, ym = make_moons(n_samples=100, noise=0.25, random_state=53) +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain the same equation as before

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ -deep_tree_clf1 = DecisionTreeClassifier(random_state=42) -deep_tree_clf2 = DecisionTreeClassifier(min_samples_leaf=4, random_state=42) -deep_tree_clf1.fit(Xm, ym) -deep_tree_clf2.fit(Xm, ym) +

    but now subject to the constraints \( \lambda_i\geq 0 \), \( \sum_i\lambda_iy_i=0 \) and \( 0\leq\lambda_i \leq C \). +We must in addition satisfy the Karush-Kuhn-Tucker condition which now reads +

    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_)\right]=0 \hspace{0.1cm}\forall i, +$$ +$$ +\gamma_i\xi_i = 0, +$$ -def plot_decision_boundary(clf, X, y, axes=[0, 7.5, 0, 3], iris=True, legend=False, plot_training=True): - x1s = np.linspace(axes[0], axes[1], 100) - x2s = np.linspace(axes[2], axes[3], 100) - x1, x2 = np.meshgrid(x1s, x2s) - X_new = np.c_[x1.ravel(), x2.ravel()] - y_pred = clf.predict(X_new).reshape(x1.shape) - custom_cmap = ListedColormap(['#fafab0','#9898ff','#a0faa0']) - plt.contourf(x1, x2, y_pred, alpha=0.3, cmap=custom_cmap) - if not iris: - custom_cmap2 = ListedColormap(['#7d7d58','#4c4c7f','#507d50']) - plt.contour(x1, x2, y_pred, cmap=custom_cmap2, alpha=0.8) - if plot_training: - plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo", label="Iris-Setosa") - plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs", label="Iris-Versicolor") - plt.plot(X[:, 0][y==2], X[:, 1][y==2], "g^", label="Iris-Virginica") - plt.axis(axes) - if iris: - plt.xlabel("Petal length", fontsize=14) - plt.ylabel("Petal width", fontsize=14) - else: - plt.xlabel(r"$x_1$", fontsize=18) - plt.ylabel(r"$x_2$", fontsize=18, rotation=0) - if legend: - plt.legend(loc="lower right", fontsize=14) -plt.figure(figsize=(11, 4)) -plt.subplot(121) -plot_decision_boundary(deep_tree_clf1, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False) -plt.title("No restrictions", fontsize=16) -plt.subplot(122) -plot_decision_boundary(deep_tree_clf2, Xm, ym, axes=[-1.5, 2.5, -1, 1.5], iris=False) -plt.title("min_samples_leaf = {}".format(deep_tree_clf2.min_samples_leaf), fontsize=14) -plt.show() -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    -
    - +

    and

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_) \geq 0 \hspace{0.1cm}\forall i. +$$

    @@ -494,18 +288,6 @@ plt.show()

  • 38
  • 39
  • 40
  • -
  • 41
  • -
  • 42
  • -
  • 43
  • -
  • 44
  • -
  • 45
  • -
  • 46
  • -
  • 47
  • -
  • 48
  • -
  • 49
  • -
  • ...
  • -
  • 79
  • -
  • »
  • diff --git a/doc/pub/week45/html/week45-bs.html b/doc/pub/week45/html/week45-bs.html index b661511d2..2de25d156 100644 --- a/doc/pub/week45/html/week45-bs.html +++ b/doc/pub/week45/html/week45-bs.html @@ -102,7 +102,43 @@ doconce format html week45.do.txt --html_style=bootstrap --pygments_html_style=d ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -158,6 +194,24 @@ MathJax.Hub.Config({
  • XGBoost: Extreme Gradient Boosting
  • Regression Case
  • Xgboost on the Cancer Data
  • +
  • Support Vector Machines, overarching aims
  • +
  • Hyperplanes and all that
  • +
  • What is a hyperplane?
  • +
  • A \( p \)-dimensional space of features
  • +
  • The two-dimensional case
  • +
  • Getting into the details
  • +
  • First attempt at a minimization approach
  • +
  • Solving the equations
  • +
  • Code Example
  • +
  • Problems with the Simpler Approach
  • +
  • A better approach
  • +
  • A quick Reminder on Lagrangian Multipliers
  • +
  • Adding the Multiplier
  • +
  • Setting up the Problem
  • +
  • The problem to solve
  • +
  • The last steps
  • +
  • A soft classifier
  • +
  • Soft optmization problem
  • @@ -212,7 +266,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 22
  • +
  • 40
  • »
  • diff --git a/doc/pub/week45/html/week45-reveal.html b/doc/pub/week45/html/week45-reveal.html index 282178247..4957afcce 100644 --- a/doc/pub/week45/html/week45-reveal.html +++ b/doc/pub/week45/html/week45-reveal.html @@ -1116,6 +1116,762 @@ plt.show()
    +
    +

    Support Vector Machines, overarching aims

    + +

    A Support Vector Machine (SVM) is a very powerful and versatile +Machine Learning method, capable of performing linear or nonlinear +classification, regression, and even outlier detection. It is one of +the most popular models in Machine Learning, and anyone interested in +Machine Learning should have it in their toolbox. SVMs are +particularly well suited for classification of complex but small-sized or +medium-sized datasets. +

    + +

    The case with two well-separated classes only can be understood in an +intuitive way in terms of lines in a two-dimensional space separating +the two classes (see figure below). +

    + +

    The basic mathematics behind the SVM is however less familiar to most of us. +It relies on the definition of hyperplanes and the +definition of a margin which separates classes (in case of +classification problems) of variables. It is also used for regression +problems. +

    + +

    With SVMs we distinguish between hard margin and soft margins. The +latter introduces a so-called softening parameter to be discussed +below. We distinguish also between linear and non-linear +approaches. The latter are the most frequent ones since it is rather +unlikely that we can separate classes easily by say straight lines. +

    +
    + +
    +

    Hyperplanes and all that

    + +

    The theory behind support vector machines (SVM hereafter) is based on +the mathematical description of so-called hyperplanes. Let us start +with a two-dimensional case. This will also allow us to introduce our +first SVM examples. These will be tailored to the case of two specific +classes, as displayed in the figure here based on the usage of the petal data. +

    + +

    We assume here that our data set can be well separated into two +domains, where a straight line does the job in the separating the two +classes. Here the two classes are represented by either squares or +circles. +

    + + +
    +
    +
    +
    +
    +
    from sklearn import datasets
    +from sklearn.svm import SVC, LinearSVC
    +from sklearn.linear_model import SGDClassifier
    +from sklearn.preprocessing import StandardScaler
    +import matplotlib
    +import matplotlib.pyplot as plt
    +plt.rcParams['axes.labelsize'] = 14
    +plt.rcParams['xtick.labelsize'] = 12
    +plt.rcParams['ytick.labelsize'] = 12
    +
    +
    +iris = datasets.load_iris()
    +X = iris["data"][:, (2, 3)]  # petal length, petal width
    +y = iris["target"]
    +
    +setosa_or_versicolor = (y == 0) | (y == 1)
    +X = X[setosa_or_versicolor]
    +y = y[setosa_or_versicolor]
    +
    +
    +
    +C = 5
    +alpha = 1 / (C * len(X))
    +
    +lin_clf = LinearSVC(loss="hinge", C=C, random_state=42)
    +svm_clf = SVC(kernel="linear", C=C)
    +sgd_clf = SGDClassifier(loss="hinge", learning_rate="constant", eta0=0.001, alpha=alpha,
    +                        max_iter=100000, random_state=42)
    +
    +scaler = StandardScaler()
    +X_scaled = scaler.fit_transform(X)
    +
    +lin_clf.fit(X_scaled, y)
    +svm_clf.fit(X_scaled, y)
    +sgd_clf.fit(X_scaled, y)
    +
    +print("LinearSVC:                   ", lin_clf.intercept_, lin_clf.coef_)
    +print("SVC:                         ", svm_clf.intercept_, svm_clf.coef_)
    +print("SGDClassifier(alpha={:.5f}):".format(sgd_clf.alpha), sgd_clf.intercept_, sgd_clf.coef_)
    +
    +# Compute the slope and bias of each decision boundary
    +w1 = -lin_clf.coef_[0, 0]/lin_clf.coef_[0, 1]
    +b1 = -lin_clf.intercept_[0]/lin_clf.coef_[0, 1]
    +w2 = -svm_clf.coef_[0, 0]/svm_clf.coef_[0, 1]
    +b2 = -svm_clf.intercept_[0]/svm_clf.coef_[0, 1]
    +w3 = -sgd_clf.coef_[0, 0]/sgd_clf.coef_[0, 1]
    +b3 = -sgd_clf.intercept_[0]/sgd_clf.coef_[0, 1]
    +
    +# Transform the decision boundary lines back to the original scale
    +line1 = scaler.inverse_transform([[-10, -10 * w1 + b1], [10, 10 * w1 + b1]])
    +line2 = scaler.inverse_transform([[-10, -10 * w2 + b2], [10, 10 * w2 + b2]])
    +line3 = scaler.inverse_transform([[-10, -10 * w3 + b3], [10, 10 * w3 + b3]])
    +
    +# Plot all three decision boundaries
    +plt.figure(figsize=(11, 4))
    +plt.plot(line1[:, 0], line1[:, 1], "k:", label="LinearSVC")
    +plt.plot(line2[:, 0], line2[:, 1], "b--", linewidth=2, label="SVC")
    +plt.plot(line3[:, 0], line3[:, 1], "r-", label="SGDClassifier")
    +plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs") # label="Iris-Versicolor"
    +plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo") # label="Iris-Setosa"
    +plt.xlabel("Petal length", fontsize=14)
    +plt.ylabel("Petal width", fontsize=14)
    +plt.legend(loc="upper center", fontsize=14)
    +plt.axis([0, 5.5, 0, 2])
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +
    +

    What is a hyperplane?

    + +

    The aim of the SVM algorithm is to find a hyperplane in a +\( p \)-dimensional space, where \( p \) is the number of features that +distinctly classifies the data points. +

    + +

    In a \( p \)-dimensional space, a hyperplane is what we call an affine subspace of dimension of \( p-1 \). +As an example, in two dimension, a hyperplane is simply as straight line while in three dimensions it is +a two-dimensional subspace, or stated simply, a plane. +

    + +

    In two dimensions, with the variables \( x_1 \) and \( x_2 \), the hyperplane is defined as

    +

     
    +$$ +b+w_1x_1+w_2x_2=0, +$$ +

     
    + +

    where \( b \) is the intercept and \( w_1 \) and \( w_2 \) define the elements of a vector orthogonal to the line +\( b+w_1x_1+w_2x_2=0 \). +In two dimensions we define the vectors \( \boldsymbol{x} =[x1,x2] \) and \( \boldsymbol{w}=[w1,w2] \). +We can then rewrite the above equation as +

    + +

     
    +$$ +\boldsymbol{x}^T\boldsymbol{w}+b=0. +$$ +

     
    +

    + +
    +

    A \( p \)-dimensional space of features

    + +

    We limit ourselves to two classes of outputs \( y_i \) and assign these classes the values \( y_i = \pm 1 \). +In a \( p \)-dimensional space of say \( p \) features we have a hyperplane defines as +

    +

     
    +$$ +b+wx_1+w_2x_2+\dots +w_px_p=0. +$$ +

     
    + +

    If we define a +matrix \( \boldsymbol{X}=\left[\boldsymbol{x}_1,\boldsymbol{x}_2,\dots, \boldsymbol{x}_p\right] \) +of dimension \( n\times p \), where \( n \) represents the observations for each feature and each vector \( x_i \) is a column vector of the matrix \( \boldsymbol{X} \), +

    +

     
    +$$ +\boldsymbol{x}_i = \begin{bmatrix} x_{i1} \\ x_{i2} \\ \dots \\ \dots \\ x_{ip} \end{bmatrix}. +$$ +

     
    + +

    If the above condition is not met for a given vector \( \boldsymbol{x}_i \) we have

    +

     
    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} >0, +$$ +

     
    + +

    if our output \( y_i=1 \). +In this case we say that \( \boldsymbol{x}_i \) lies on one of the sides of the hyperplane and if +

    +

     
    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} < 0, +$$ +

     
    + +

    for the class of observations \( y_i=-1 \), +then \( \boldsymbol{x}_i \) lies on the other side. +

    + +

    Equivalently, for the two classes of observations we have

    +

     
    +$$ +y_i\left(b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip}\right) > 0. +$$ +

     
    + +

    When we try to separate hyperplanes, if it exists, we can use it to construct a natural classifier: a test observation is assigned a given class depending on which side of the hyperplane it is located.

    +
    + +
    +

    The two-dimensional case

    + +

    Let us try to develop our intuition about SVMs by limiting ourselves to a two-dimensional +plane. To separate the two classes of data points, there are many +possible lines (hyperplanes if you prefer a more strict naming) +that could be chosen. Our objective is to find a +plane that has the maximum margin, i.e the maximum distance between +data points of both classes. Maximizing the margin distance provides +some reinforcement so that future data points can be classified with +more confidence. +

    + +

    What a linear classifier attempts to accomplish is to split the +feature space into two half spaces by placing a hyperplane between the +data points. This hyperplane will be our decision boundary. All +points on one side of the plane will belong to class one and all points +on the other side of the plane will belong to the second class two. +

    + +

    Unfortunately there are many ways in which we can place a hyperplane +to divide the data. Below is an example of two candidate hyperplanes +for our data sample. +

    +
    + +
    +

    Getting into the details

    + +

    Let us define the function

    +

     
    +$$ +f(x) = \boldsymbol{w}^T\boldsymbol{x}+b = 0, +$$ +

     
    + +

    as the function that determines the line \( L \) that separates two classes (our two features), see the figure here.

    + +

    Any point defined by \( \boldsymbol{x}_i \) and \( \boldsymbol{x}_2 \) on the line \( L \) will satisfy \( \boldsymbol{w}^T(\boldsymbol{x}_1-\boldsymbol{x}_2)=0 \).

    + +

    The signed distance \( \delta \) from any point defined by a vector \( \boldsymbol{x} \) and a point \( \boldsymbol{x}_0 \) on the line \( L \) is then

    +

     
    +$$ +\delta = \frac{1}{\vert\vert \boldsymbol{w}\vert\vert}(\boldsymbol{w}^T\boldsymbol{x}+b). +$$ +

     
    +

    + +
    +

    First attempt at a minimization approach

    + +

    How do we find the parameter \( b \) and the vector \( \boldsymbol{w} \)? What we could +do is to define a cost function which now contains the set of all +misclassified points \( M \) and attempt to minimize this function +

    + +

     
    +$$ +C(\boldsymbol{w},b) = -\sum_{i\in M} y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ +

     
    + +

    We could now for example define all values \( y_i =1 \) as misclassified in case we have \( \boldsymbol{w}^T\boldsymbol{x}_i+b < 0 \) and the opposite if we have \( y_i=-1 \). Taking the derivatives gives us

    +

     
    +$$ +\frac{\partial C}{\partial b} = -\sum_{i\in M} y_i, +$$ +

     
    + +

    and

    +

     
    +$$ +\frac{\partial C}{\partial \boldsymbol{w}} = -\sum_{i\in M} y_ix_i. +$$ +

     
    +

    + +
    +

    Solving the equations

    + +

    We can now use the Newton-Raphson method or different variants of the gradient descent family (from plain gradient descent to various stochastic gradient descent approaches) to solve the equations

    +

     
    +$$ +b \leftarrow b +\eta \frac{\partial C}{\partial b}, +$$ +

     
    + +

    and

    +

     
    +$$ +\boldsymbol{w} \leftarrow \boldsymbol{w} +\eta \frac{\partial C}{\partial \boldsymbol{w}}, +$$ +

     
    + +

    where \( \eta \) is our by now well-known learning rate.

    +
    + +
    +

    Code Example

    + +

    The equations we discussed above can be coded rather easily (the +framework is similar to what we developed for logistic +regression). We are going to set up a simple case with two classes only and we want to find a line which separates them the best possible way. +

    + + +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + +
    +

    Problems with the Simpler Approach

    + +

    There are however problems with this approach, although it looks +pretty straightforward to implement. When running the above code, we see that we can easily end up with many diffeent lines which separate the two classes. +

    + +

    For small +gaps between the entries, we may also end up needing many iterations +before the solutions converge and if the data cannot be separated +properly into two distinct classes, we may not experience a converge +at all. +

    +
    + +
    +

    A better approach

    + +

    A better approach is rather to try to define a large margin between +the two classes (if they are well separated from the beginning). +

    + +

    Thus, we wish to find a margin \( M \) with \( \boldsymbol{w} \) normalized to +\( \vert\vert \boldsymbol{w}\vert\vert =1 \) subject to the condition +

    + +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, p. +$$ +

     
    + +

    All points are thus at a signed distance from the decision boundary defined by the line \( L \). The parameters \( b \) and \( w_1 \) and \( w_2 \) define this line.

    + +

    We seek thus the largest value \( M \) defined by

    +

     
    +$$ +\frac{1}{\vert \vert \boldsymbol{w}\vert\vert}y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, n, +$$ +

     
    + +

    or just

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M\vert \vert \boldsymbol{w}\vert\vert \hspace{0.1cm}\forall i. +$$ +

     
    + +

    If we scale the equation so that \( \vert \vert \boldsymbol{w}\vert\vert = 1/M \), we have to find the minimum of +\( \boldsymbol{w}^T\boldsymbol{w}=\vert \vert \boldsymbol{w}\vert\vert \) (the norm) subject to the condition +

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq 1 \hspace{0.1cm}\forall i. +$$ +

     
    + +

    We have thus defined our margin as the invers of the norm of +\( \boldsymbol{w} \). We want to minimize the norm in order to have a as large as +possible margin \( M \). Before we proceed, we need to remind ourselves +about Lagrangian multipliers. +

    +
    + +
    +

    A quick Reminder on Lagrangian Multipliers

    + +

    Consider a function of three independent variables \( f(x,y,z) \) . For the function \( f \) to be an +extreme we have +

    +

     
    +$$ +df=0. +$$ +

     
    + +

    A necessary and sufficient condition is

    +

     
    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ +

     
    + +

    due to

    +

     
    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz. +$$ +

     
    + +

    In many problems the variables \( x,y,z \) are often subject to constraints (such as those above for the margin) +so that they are no longer all independent. It is possible at least in principle to use each +constraint to eliminate one variable +and to proceed with a new and smaller set of independent varables. +

    + +

    The use of so-called Lagrangian multipliers is an alternative technique when the elimination +of variables is incovenient or undesirable. Assume that we have an equation of constraint on +the variables \( x,y,z \) +

    +

     
    +$$ +\phi(x,y,z) = 0, +$$ +

     
    + +

    resulting in

    +

     
    +$$ +d\phi = \frac{\partial \phi}{\partial x}dx+\frac{\partial \phi}{\partial y}dy+\frac{\partial \phi}{\partial z}dz =0. +$$ +

     
    + +

    Now we cannot set anymore

    +

     
    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ +

     
    + +

    if \( df=0 \) is wanted +because there are now only two independent variables! Assume \( x \) and \( y \) are the independent +variables. +Then \( dz \) is no longer arbitrary. +

    +
    + +
    +

    Adding the Multiplier

    + +

    However, we can add to

    +

     
    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz, +$$ +

     
    + +

    a multiplum of \( d\phi \), viz. \( \lambda d\phi \), resulting in

    +

     
    +$$ +df+\lambda d\phi = (\frac{\partial f}{\partial z}+\lambda +\frac{\partial \phi}{\partial x})dx+(\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y})dy+ +(\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z})dz =0. +$$ +

     
    + +

    Our multiplier is chosen so that

    +

     
    +$$ +\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z} =0. +$$ +

     
    + +

    We need to remember that we took \( dx \) and \( dy \) to be arbitrary and thus we must have

    +

     
    +$$ +\frac{\partial f}{\partial x}+\lambda\frac{\partial \phi}{\partial x} =0, +$$ +

     
    + +

    and

    +

     
    +$$ +\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y} =0. +$$ +

     
    + +

    When all these equations are satisfied, \( df=0 \). We have four unknowns, \( x,y,z \) and +\( \lambda \). Actually we want only \( x,y,z \), \( \lambda \) needs not to be determined, +it is therefore often called +Lagrange's undetermined multiplier. +If we have a set of constraints \( \phi_k \) we have the equations +

    +

     
    +$$ +\frac{\partial f}{\partial x_i}+\sum_k\lambda_k\frac{\partial \phi_k}{\partial x_i} =0. +$$ +

     
    +

    + +
    +

    Setting up the Problem

    +

    In order to solve the above problem, we define the following Lagrangian function to be minimized

    +

     
    +$$ +{\cal L}(\lambda,b,\boldsymbol{w})=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-1\right], +$$ +

     
    + +

    where \( \lambda_i \) is a so-called Lagrange multiplier subject to the condition \( \lambda_i \geq 0 \).

    + +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +

     
    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ +

     
    + +

    and

    +

     
    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ +

     
    + +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain

    +

     
    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ +

     
    + +

    subject to the constraints \( \lambda_i\geq 0 \) and \( \sum_i\lambda_iy_i=0 \). +We must in addition satisfy the Karush-Kuhn-Tucker (KKT) condition +

    +

     
    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -1\right] \hspace{0.1cm}\forall i. +$$ +

     
    + +

      +

    1. If \( \lambda_i > 0 \), then \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) and we say that \( x_i \) is on the boundary.
    2. +

    3. If \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)> 1 \), we say \( x_i \) is not on the boundary and we set \( \lambda_i=0 \).
    4. +
    +

    +

    When \( \lambda_i > 0 \), the vectors \( \boldsymbol{x}_i \) are called support vectors. They are the vectors closest to the line (or hyperplane) and define the margin \( M \).

    +
    + +
    +

    The problem to solve

    + +

    We can rewrite

    +

     
    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ +

     
    + +

    and its constraints in terms of a matrix-vector problem where we minimize w.r.t. \( \lambda \) the following problem

    +

     
    +$$ +\frac{1}{2} \boldsymbol{\lambda}^T\begin{bmatrix} y_1y_1\boldsymbol{x}_1^T\boldsymbol{x}_1 & y_1y_2\boldsymbol{x}_1^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_1^T\boldsymbol{x}_n \\ +y_2y_1\boldsymbol{x}_2^T\boldsymbol{x}_1 & y_2y_2\boldsymbol{x}_2^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_2^T\boldsymbol{x}_n \\ +\dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots \\ +y_ny_1\boldsymbol{x}_n^T\boldsymbol{x}_1 & y_ny_2\boldsymbol{x}_n^T\boldsymbol{x}_2 & \dots & \dots & y_ny_n\boldsymbol{x}_n^T\boldsymbol{x}_n \\ +\end{bmatrix}\boldsymbol{\lambda}-\mathbb{1}\boldsymbol{\lambda}, +$$ +

     
    + +

    subject to \( \boldsymbol{y}^T\boldsymbol{\lambda}=0 \). Here we defined the vectors \( \boldsymbol{\lambda} =[\lambda_1,\lambda_2,\dots,\lambda_n] \) and +\( \boldsymbol{y}=[y_1,y_2,\dots,y_n] \). +

    +
    + +
    +

    The last steps

    + +

    Solving the above problem, yields the values of \( \lambda_i \). +To find the coefficients of your hyperplane we need simply to compute +

    +

     
    +$$ +\boldsymbol{w}=\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ +

     
    + +

    With our vector \( \boldsymbol{w} \) we can in turn find the value of the intercept \( b \) (here in two dimensions) via

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ +

     
    + +

    resulting in

    +

     
    +$$ +b = \frac{1}{y_i}-\boldsymbol{w}^T\boldsymbol{x}_i, +$$ +

     
    + +

    or if we write it out in terms of the support vectors only, with \( N_s \) being their number, we have

    +

     
    +$$ +b = \frac{1}{N_s}\sum_{j\in N_s}\left(y_j-\sum_{i=1}^n\lambda_iy_i\boldsymbol{x}_i^T\boldsymbol{x}_j\right). +$$ +

     
    + +

    With our hyperplane coefficients we can use our classifier to assign any observation by simply using

    +

     
    +$$ +y_i = \mathrm{sign}(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ +

     
    + +

    Below we discuss how to find the optimal values of \( \lambda_i \). Before we proceed however, we discuss now the so-called soft classifier.

    +
    + +
    +

    A soft classifier

    + +

    Till now, the margin is strictly defined by the support vectors. This defines what is called a hard classifier, that is the margins are well defined.

    + +

    Suppose now that classes overlap in feature space, as shown in the +figure here. One way to deal with this problem before we define the +so-called kernel approach, is to allow a kind of slack in the sense +that we allow some points to be on the wrong side of the margin. +

    + +

    We introduce thus the so-called slack variables \( \boldsymbol{\xi} =[\xi_1,x_2,\dots,x_n] \) and +modify our previous equation +

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ +

     
    + +

    to

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i, +$$ +

     
    + +

    with the requirement \( \xi_i\geq 0 \). The total violation is now \( \sum_i\xi \). +The value \( \xi_i \) in the constraint the last constraint corresponds to the amount by which the prediction +\( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) is on the wrong side of its margin. Hence by bounding the sum \( \sum_i \xi_i \), +we bound the total amount by which predictions fall on the wrong side of their margins. +

    + +

    Misclassifications occur when \( \xi_i > 1 \). Thus bounding the total sum by some value \( C \) bounds in turn the total number of +misclassifications. +

    +
    + +
    +

    Soft optmization problem

    + +

    This has in turn the consequences that we change our optmization problem to finding the minimum of

    +

     
    +$$ +{\cal L}=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-(1-\xi_)\right]+C\sum_{i=1}^n\xi_i-\sum_{i=1}^n\gamma_i\xi_i, +$$ +

     
    + +

    subject to

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i \hspace{0.1cm}\forall i, +$$ +

     
    + +

    with the requirement \( \xi_i\geq 0 \).

    + +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +

     
    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ +

     
    + +

    and

    +

     
    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i, +$$ +

     
    + +

    and

    +

     
    +$$ +\lambda_i = C-\gamma_i \hspace{0.1cm}\forall i. +$$ +

     
    + +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain the same equation as before

    +

     
    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ +

     
    + +

    but now subject to the constraints \( \lambda_i\geq 0 \), \( \sum_i\lambda_iy_i=0 \) and \( 0\leq\lambda_i \leq C \). +We must in addition satisfy the Karush-Kuhn-Tucker condition which now reads +

    +

     
    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_)\right]=0 \hspace{0.1cm}\forall i, +$$ +

     
    + +

     
    +$$ +\gamma_i\xi_i = 0, +$$ +

     
    + +

    and

    +

     
    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_) \geq 0 \hspace{0.1cm}\forall i. +$$ +

     
    +

    +
    diff --git a/doc/pub/week45/html/week45-solarized.html b/doc/pub/week45/html/week45-solarized.html index 014adef21..6f0c995d0 100644 --- a/doc/pub/week45/html/week45-solarized.html +++ b/doc/pub/week45/html/week45-solarized.html @@ -129,7 +129,43 @@ div.toc p,a { ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -1010,6 +1046,643 @@ plt.show()
    +









    +

    Support Vector Machines, overarching aims

    + +

    A Support Vector Machine (SVM) is a very powerful and versatile +Machine Learning method, capable of performing linear or nonlinear +classification, regression, and even outlier detection. It is one of +the most popular models in Machine Learning, and anyone interested in +Machine Learning should have it in their toolbox. SVMs are +particularly well suited for classification of complex but small-sized or +medium-sized datasets. +

    + +

    The case with two well-separated classes only can be understood in an +intuitive way in terms of lines in a two-dimensional space separating +the two classes (see figure below). +

    + +

    The basic mathematics behind the SVM is however less familiar to most of us. +It relies on the definition of hyperplanes and the +definition of a margin which separates classes (in case of +classification problems) of variables. It is also used for regression +problems. +

    + +

    With SVMs we distinguish between hard margin and soft margins. The +latter introduces a so-called softening parameter to be discussed +below. We distinguish also between linear and non-linear +approaches. The latter are the most frequent ones since it is rather +unlikely that we can separate classes easily by say straight lines. +

    + +









    +

    Hyperplanes and all that

    + +

    The theory behind support vector machines (SVM hereafter) is based on +the mathematical description of so-called hyperplanes. Let us start +with a two-dimensional case. This will also allow us to introduce our +first SVM examples. These will be tailored to the case of two specific +classes, as displayed in the figure here based on the usage of the petal data. +

    + +

    We assume here that our data set can be well separated into two +domains, where a straight line does the job in the separating the two +classes. Here the two classes are represented by either squares or +circles. +

    + + +
    +
    +
    +
    +
    +
    from sklearn import datasets
    +from sklearn.svm import SVC, LinearSVC
    +from sklearn.linear_model import SGDClassifier
    +from sklearn.preprocessing import StandardScaler
    +import matplotlib
    +import matplotlib.pyplot as plt
    +plt.rcParams['axes.labelsize'] = 14
    +plt.rcParams['xtick.labelsize'] = 12
    +plt.rcParams['ytick.labelsize'] = 12
    +
    +
    +iris = datasets.load_iris()
    +X = iris["data"][:, (2, 3)]  # petal length, petal width
    +y = iris["target"]
    +
    +setosa_or_versicolor = (y == 0) | (y == 1)
    +X = X[setosa_or_versicolor]
    +y = y[setosa_or_versicolor]
    +
    +
    +
    +C = 5
    +alpha = 1 / (C * len(X))
    +
    +lin_clf = LinearSVC(loss="hinge", C=C, random_state=42)
    +svm_clf = SVC(kernel="linear", C=C)
    +sgd_clf = SGDClassifier(loss="hinge", learning_rate="constant", eta0=0.001, alpha=alpha,
    +                        max_iter=100000, random_state=42)
    +
    +scaler = StandardScaler()
    +X_scaled = scaler.fit_transform(X)
    +
    +lin_clf.fit(X_scaled, y)
    +svm_clf.fit(X_scaled, y)
    +sgd_clf.fit(X_scaled, y)
    +
    +print("LinearSVC:                   ", lin_clf.intercept_, lin_clf.coef_)
    +print("SVC:                         ", svm_clf.intercept_, svm_clf.coef_)
    +print("SGDClassifier(alpha={:.5f}):".format(sgd_clf.alpha), sgd_clf.intercept_, sgd_clf.coef_)
    +
    +# Compute the slope and bias of each decision boundary
    +w1 = -lin_clf.coef_[0, 0]/lin_clf.coef_[0, 1]
    +b1 = -lin_clf.intercept_[0]/lin_clf.coef_[0, 1]
    +w2 = -svm_clf.coef_[0, 0]/svm_clf.coef_[0, 1]
    +b2 = -svm_clf.intercept_[0]/svm_clf.coef_[0, 1]
    +w3 = -sgd_clf.coef_[0, 0]/sgd_clf.coef_[0, 1]
    +b3 = -sgd_clf.intercept_[0]/sgd_clf.coef_[0, 1]
    +
    +# Transform the decision boundary lines back to the original scale
    +line1 = scaler.inverse_transform([[-10, -10 * w1 + b1], [10, 10 * w1 + b1]])
    +line2 = scaler.inverse_transform([[-10, -10 * w2 + b2], [10, 10 * w2 + b2]])
    +line3 = scaler.inverse_transform([[-10, -10 * w3 + b3], [10, 10 * w3 + b3]])
    +
    +# Plot all three decision boundaries
    +plt.figure(figsize=(11, 4))
    +plt.plot(line1[:, 0], line1[:, 1], "k:", label="LinearSVC")
    +plt.plot(line2[:, 0], line2[:, 1], "b--", linewidth=2, label="SVC")
    +plt.plot(line3[:, 0], line3[:, 1], "r-", label="SGDClassifier")
    +plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs") # label="Iris-Versicolor"
    +plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo") # label="Iris-Setosa"
    +plt.xlabel("Petal length", fontsize=14)
    +plt.ylabel("Petal width", fontsize=14)
    +plt.legend(loc="upper center", fontsize=14)
    +plt.axis([0, 5.5, 0, 2])
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + + +









    +

    What is a hyperplane?

    + +

    The aim of the SVM algorithm is to find a hyperplane in a +\( p \)-dimensional space, where \( p \) is the number of features that +distinctly classifies the data points. +

    + +

    In a \( p \)-dimensional space, a hyperplane is what we call an affine subspace of dimension of \( p-1 \). +As an example, in two dimension, a hyperplane is simply as straight line while in three dimensions it is +a two-dimensional subspace, or stated simply, a plane. +

    + +

    In two dimensions, with the variables \( x_1 \) and \( x_2 \), the hyperplane is defined as

    +$$ +b+w_1x_1+w_2x_2=0, +$$ + +

    where \( b \) is the intercept and \( w_1 \) and \( w_2 \) define the elements of a vector orthogonal to the line +\( b+w_1x_1+w_2x_2=0 \). +In two dimensions we define the vectors \( \boldsymbol{x} =[x1,x2] \) and \( \boldsymbol{w}=[w1,w2] \). +We can then rewrite the above equation as +

    + +$$ +\boldsymbol{x}^T\boldsymbol{w}+b=0. +$$ + + +









    +

    A \( p \)-dimensional space of features

    + +

    We limit ourselves to two classes of outputs \( y_i \) and assign these classes the values \( y_i = \pm 1 \). +In a \( p \)-dimensional space of say \( p \) features we have a hyperplane defines as +

    +$$ +b+wx_1+w_2x_2+\dots +w_px_p=0. +$$ + +

    If we define a +matrix \( \boldsymbol{X}=\left[\boldsymbol{x}_1,\boldsymbol{x}_2,\dots, \boldsymbol{x}_p\right] \) +of dimension \( n\times p \), where \( n \) represents the observations for each feature and each vector \( x_i \) is a column vector of the matrix \( \boldsymbol{X} \), +

    +$$ +\boldsymbol{x}_i = \begin{bmatrix} x_{i1} \\ x_{i2} \\ \dots \\ \dots \\ x_{ip} \end{bmatrix}. +$$ + +

    If the above condition is not met for a given vector \( \boldsymbol{x}_i \) we have

    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} >0, +$$ + +

    if our output \( y_i=1 \). +In this case we say that \( \boldsymbol{x}_i \) lies on one of the sides of the hyperplane and if +

    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} < 0, +$$ + +

    for the class of observations \( y_i=-1 \), +then \( \boldsymbol{x}_i \) lies on the other side. +

    + +

    Equivalently, for the two classes of observations we have

    +$$ +y_i\left(b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip}\right) > 0. +$$ + +

    When we try to separate hyperplanes, if it exists, we can use it to construct a natural classifier: a test observation is assigned a given class depending on which side of the hyperplane it is located.

    + + +

    The two-dimensional case

    + +

    Let us try to develop our intuition about SVMs by limiting ourselves to a two-dimensional +plane. To separate the two classes of data points, there are many +possible lines (hyperplanes if you prefer a more strict naming) +that could be chosen. Our objective is to find a +plane that has the maximum margin, i.e the maximum distance between +data points of both classes. Maximizing the margin distance provides +some reinforcement so that future data points can be classified with +more confidence. +

    + +

    What a linear classifier attempts to accomplish is to split the +feature space into two half spaces by placing a hyperplane between the +data points. This hyperplane will be our decision boundary. All +points on one side of the plane will belong to class one and all points +on the other side of the plane will belong to the second class two. +

    + +

    Unfortunately there are many ways in which we can place a hyperplane +to divide the data. Below is an example of two candidate hyperplanes +for our data sample. +

    + +









    +

    Getting into the details

    + +

    Let us define the function

    +$$ +f(x) = \boldsymbol{w}^T\boldsymbol{x}+b = 0, +$$ + +

    as the function that determines the line \( L \) that separates two classes (our two features), see the figure here.

    + +

    Any point defined by \( \boldsymbol{x}_i \) and \( \boldsymbol{x}_2 \) on the line \( L \) will satisfy \( \boldsymbol{w}^T(\boldsymbol{x}_1-\boldsymbol{x}_2)=0 \).

    + +

    The signed distance \( \delta \) from any point defined by a vector \( \boldsymbol{x} \) and a point \( \boldsymbol{x}_0 \) on the line \( L \) is then

    +$$ +\delta = \frac{1}{\vert\vert \boldsymbol{w}\vert\vert}(\boldsymbol{w}^T\boldsymbol{x}+b). +$$ + + +









    +

    First attempt at a minimization approach

    + +

    How do we find the parameter \( b \) and the vector \( \boldsymbol{w} \)? What we could +do is to define a cost function which now contains the set of all +misclassified points \( M \) and attempt to minimize this function +

    + +$$ +C(\boldsymbol{w},b) = -\sum_{i\in M} y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ + +

    We could now for example define all values \( y_i =1 \) as misclassified in case we have \( \boldsymbol{w}^T\boldsymbol{x}_i+b < 0 \) and the opposite if we have \( y_i=-1 \). Taking the derivatives gives us

    +$$ +\frac{\partial C}{\partial b} = -\sum_{i\in M} y_i, +$$ + +

    and

    +$$ +\frac{\partial C}{\partial \boldsymbol{w}} = -\sum_{i\in M} y_ix_i. +$$ + + +









    +

    Solving the equations

    + +

    We can now use the Newton-Raphson method or different variants of the gradient descent family (from plain gradient descent to various stochastic gradient descent approaches) to solve the equations

    +$$ +b \leftarrow b +\eta \frac{\partial C}{\partial b}, +$$ + +

    and

    +$$ +\boldsymbol{w} \leftarrow \boldsymbol{w} +\eta \frac{\partial C}{\partial \boldsymbol{w}}, +$$ + +

    where \( \eta \) is our by now well-known learning rate.

    + +









    +

    Code Example

    + +

    The equations we discussed above can be coded rather easily (the +framework is similar to what we developed for logistic +regression). We are going to set up a simple case with two classes only and we want to find a line which separates them the best possible way. +

    + + +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + + +









    +

    Problems with the Simpler Approach

    + +

    There are however problems with this approach, although it looks +pretty straightforward to implement. When running the above code, we see that we can easily end up with many diffeent lines which separate the two classes. +

    + +

    For small +gaps between the entries, we may also end up needing many iterations +before the solutions converge and if the data cannot be separated +properly into two distinct classes, we may not experience a converge +at all. +

    + +









    +

    A better approach

    + +

    A better approach is rather to try to define a large margin between +the two classes (if they are well separated from the beginning). +

    + +

    Thus, we wish to find a margin \( M \) with \( \boldsymbol{w} \) normalized to +\( \vert\vert \boldsymbol{w}\vert\vert =1 \) subject to the condition +

    + +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, p. +$$ + +

    All points are thus at a signed distance from the decision boundary defined by the line \( L \). The parameters \( b \) and \( w_1 \) and \( w_2 \) define this line.

    + +

    We seek thus the largest value \( M \) defined by

    +$$ +\frac{1}{\vert \vert \boldsymbol{w}\vert\vert}y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, n, +$$ + +

    or just

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M\vert \vert \boldsymbol{w}\vert\vert \hspace{0.1cm}\forall i. +$$ + +

    If we scale the equation so that \( \vert \vert \boldsymbol{w}\vert\vert = 1/M \), we have to find the minimum of +\( \boldsymbol{w}^T\boldsymbol{w}=\vert \vert \boldsymbol{w}\vert\vert \) (the norm) subject to the condition +

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq 1 \hspace{0.1cm}\forall i. +$$ + +

    We have thus defined our margin as the invers of the norm of +\( \boldsymbol{w} \). We want to minimize the norm in order to have a as large as +possible margin \( M \). Before we proceed, we need to remind ourselves +about Lagrangian multipliers. +

    + +









    +

    A quick Reminder on Lagrangian Multipliers

    + +

    Consider a function of three independent variables \( f(x,y,z) \) . For the function \( f \) to be an +extreme we have +

    +$$ +df=0. +$$ + +

    A necessary and sufficient condition is

    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ + +

    due to

    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz. +$$ + +

    In many problems the variables \( x,y,z \) are often subject to constraints (such as those above for the margin) +so that they are no longer all independent. It is possible at least in principle to use each +constraint to eliminate one variable +and to proceed with a new and smaller set of independent varables. +

    + +

    The use of so-called Lagrangian multipliers is an alternative technique when the elimination +of variables is incovenient or undesirable. Assume that we have an equation of constraint on +the variables \( x,y,z \) +

    +$$ +\phi(x,y,z) = 0, +$$ + +

    resulting in

    +$$ +d\phi = \frac{\partial \phi}{\partial x}dx+\frac{\partial \phi}{\partial y}dy+\frac{\partial \phi}{\partial z}dz =0. +$$ + +

    Now we cannot set anymore

    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ + +

    if \( df=0 \) is wanted +because there are now only two independent variables! Assume \( x \) and \( y \) are the independent +variables. +Then \( dz \) is no longer arbitrary. +

    + +









    +

    Adding the Multiplier

    + +

    However, we can add to

    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz, +$$ + +

    a multiplum of \( d\phi \), viz. \( \lambda d\phi \), resulting in

    +$$ +df+\lambda d\phi = (\frac{\partial f}{\partial z}+\lambda +\frac{\partial \phi}{\partial x})dx+(\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y})dy+ +(\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z})dz =0. +$$ + +

    Our multiplier is chosen so that

    +$$ +\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z} =0. +$$ + +

    We need to remember that we took \( dx \) and \( dy \) to be arbitrary and thus we must have

    +$$ +\frac{\partial f}{\partial x}+\lambda\frac{\partial \phi}{\partial x} =0, +$$ + +

    and

    +$$ +\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y} =0. +$$ + +

    When all these equations are satisfied, \( df=0 \). We have four unknowns, \( x,y,z \) and +\( \lambda \). Actually we want only \( x,y,z \), \( \lambda \) needs not to be determined, +it is therefore often called +Lagrange's undetermined multiplier. +If we have a set of constraints \( \phi_k \) we have the equations +

    +$$ +\frac{\partial f}{\partial x_i}+\sum_k\lambda_k\frac{\partial \phi_k}{\partial x_i} =0. +$$ + + +









    +

    Setting up the Problem

    +

    In order to solve the above problem, we define the following Lagrangian function to be minimized

    +$$ +{\cal L}(\lambda,b,\boldsymbol{w})=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-1\right], +$$ + +

    where \( \lambda_i \) is a so-called Lagrange multiplier subject to the condition \( \lambda_i \geq 0 \).

    + +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ + +

    and

    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ + +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    subject to the constraints \( \lambda_i\geq 0 \) and \( \sum_i\lambda_iy_i=0 \). +We must in addition satisfy the Karush-Kuhn-Tucker (KKT) condition +

    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -1\right] \hspace{0.1cm}\forall i. +$$ + +
      +
    1. If \( \lambda_i > 0 \), then \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) and we say that \( x_i \) is on the boundary.
    2. +
    3. If \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)> 1 \), we say \( x_i \) is not on the boundary and we set \( \lambda_i=0 \).
    4. +
    +

    When \( \lambda_i > 0 \), the vectors \( \boldsymbol{x}_i \) are called support vectors. They are the vectors closest to the line (or hyperplane) and define the margin \( M \).

    + +









    +

    The problem to solve

    + +

    We can rewrite

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    and its constraints in terms of a matrix-vector problem where we minimize w.r.t. \( \lambda \) the following problem

    +$$ +\frac{1}{2} \boldsymbol{\lambda}^T\begin{bmatrix} y_1y_1\boldsymbol{x}_1^T\boldsymbol{x}_1 & y_1y_2\boldsymbol{x}_1^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_1^T\boldsymbol{x}_n \\ +y_2y_1\boldsymbol{x}_2^T\boldsymbol{x}_1 & y_2y_2\boldsymbol{x}_2^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_2^T\boldsymbol{x}_n \\ +\dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots \\ +y_ny_1\boldsymbol{x}_n^T\boldsymbol{x}_1 & y_ny_2\boldsymbol{x}_n^T\boldsymbol{x}_2 & \dots & \dots & y_ny_n\boldsymbol{x}_n^T\boldsymbol{x}_n \\ +\end{bmatrix}\boldsymbol{\lambda}-\mathbb{1}\boldsymbol{\lambda}, +$$ + +

    subject to \( \boldsymbol{y}^T\boldsymbol{\lambda}=0 \). Here we defined the vectors \( \boldsymbol{\lambda} =[\lambda_1,\lambda_2,\dots,\lambda_n] \) and +\( \boldsymbol{y}=[y_1,y_2,\dots,y_n] \). +

    + +









    +

    The last steps

    + +

    Solving the above problem, yields the values of \( \lambda_i \). +To find the coefficients of your hyperplane we need simply to compute +

    +$$ +\boldsymbol{w}=\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ + +

    With our vector \( \boldsymbol{w} \) we can in turn find the value of the intercept \( b \) (here in two dimensions) via

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ + +

    resulting in

    +$$ +b = \frac{1}{y_i}-\boldsymbol{w}^T\boldsymbol{x}_i, +$$ + +

    or if we write it out in terms of the support vectors only, with \( N_s \) being their number, we have

    +$$ +b = \frac{1}{N_s}\sum_{j\in N_s}\left(y_j-\sum_{i=1}^n\lambda_iy_i\boldsymbol{x}_i^T\boldsymbol{x}_j\right). +$$ + +

    With our hyperplane coefficients we can use our classifier to assign any observation by simply using

    +$$ +y_i = \mathrm{sign}(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ + +

    Below we discuss how to find the optimal values of \( \lambda_i \). Before we proceed however, we discuss now the so-called soft classifier.

    + +









    +

    A soft classifier

    + +

    Till now, the margin is strictly defined by the support vectors. This defines what is called a hard classifier, that is the margins are well defined.

    + +

    Suppose now that classes overlap in feature space, as shown in the +figure here. One way to deal with this problem before we define the +so-called kernel approach, is to allow a kind of slack in the sense +that we allow some points to be on the wrong side of the margin. +

    + +

    We introduce thus the so-called slack variables \( \boldsymbol{\xi} =[\xi_1,x_2,\dots,x_n] \) and +modify our previous equation +

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ + +

    to

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i, +$$ + +

    with the requirement \( \xi_i\geq 0 \). The total violation is now \( \sum_i\xi \). +The value \( \xi_i \) in the constraint the last constraint corresponds to the amount by which the prediction +\( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) is on the wrong side of its margin. Hence by bounding the sum \( \sum_i \xi_i \), +we bound the total amount by which predictions fall on the wrong side of their margins. +

    + +

    Misclassifications occur when \( \xi_i > 1 \). Thus bounding the total sum by some value \( C \) bounds in turn the total number of +misclassifications. +

    + +









    +

    Soft optmization problem

    + +

    This has in turn the consequences that we change our optmization problem to finding the minimum of

    +$$ +{\cal L}=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-(1-\xi_)\right]+C\sum_{i=1}^n\xi_i-\sum_{i=1}^n\gamma_i\xi_i, +$$ + +

    subject to

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i \hspace{0.1cm}\forall i, +$$ + +

    with the requirement \( \xi_i\geq 0 \).

    + +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ + +

    and

    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i, +$$ + +

    and

    +$$ +\lambda_i = C-\gamma_i \hspace{0.1cm}\forall i. +$$ + +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain the same equation as before

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    but now subject to the constraints \( \lambda_i\geq 0 \), \( \sum_i\lambda_iy_i=0 \) and \( 0\leq\lambda_i \leq C \). +We must in addition satisfy the Karush-Kuhn-Tucker condition which now reads +

    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_)\right]=0 \hspace{0.1cm}\forall i, +$$ + +$$ +\gamma_i\xi_i = 0, +$$ + +

    and

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_) \geq 0 \hspace{0.1cm}\forall i. +$$ +
    © 1999-2022, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license diff --git a/doc/pub/week45/html/week45.html b/doc/pub/week45/html/week45.html index 319ea44fe..57046d980 100644 --- a/doc/pub/week45/html/week45.html +++ b/doc/pub/week45/html/week45.html @@ -206,7 +206,43 @@ div.toc p,a { ('Xgboost on the Cancer Data', 2, None, - 'xgboost-on-the-cancer-data')]} + 'xgboost-on-the-cancer-data'), + ('Support Vector Machines, overarching aims', + 2, + None, + 'support-vector-machines-overarching-aims'), + ('Hyperplanes and all that', 2, None, 'hyperplanes-and-all-that'), + ('What is a hyperplane?', 2, None, 'what-is-a-hyperplane'), + ('A $p$-dimensional space of features', + 2, + None, + 'a-p-dimensional-space-of-features'), + ('The two-dimensional case', 2, None, 'the-two-dimensional-case'), + ('Getting into the details', 2, None, 'getting-into-the-details'), + ('First attempt at a minimization approach', + 2, + None, + 'first-attempt-at-a-minimization-approach'), + ('Solving the equations', 2, None, 'solving-the-equations'), + ('Code Example', 2, None, 'code-example'), + ('Problems with the Simpler Approach', + 2, + None, + 'problems-with-the-simpler-approach'), + ('A better approach', 2, None, 'a-better-approach'), + ('A quick Reminder on Lagrangian Multipliers', + 2, + None, + 'a-quick-reminder-on-lagrangian-multipliers'), + ('Adding the Multiplier', 2, None, 'adding-the-multiplier'), + ('Setting up the Problem', 2, None, 'setting-up-the-problem'), + ('The problem to solve', 2, None, 'the-problem-to-solve'), + ('The last steps', 2, None, 'the-last-steps'), + ('A soft classifier', 2, None, 'a-soft-classifier'), + ('Soft optmization problem', + 2, + None, + 'soft-optmization-problem')]} end of tocinfo --> @@ -1087,6 +1123,643 @@ plt.show()
    +









    +

    Support Vector Machines, overarching aims

    + +

    A Support Vector Machine (SVM) is a very powerful and versatile +Machine Learning method, capable of performing linear or nonlinear +classification, regression, and even outlier detection. It is one of +the most popular models in Machine Learning, and anyone interested in +Machine Learning should have it in their toolbox. SVMs are +particularly well suited for classification of complex but small-sized or +medium-sized datasets. +

    + +

    The case with two well-separated classes only can be understood in an +intuitive way in terms of lines in a two-dimensional space separating +the two classes (see figure below). +

    + +

    The basic mathematics behind the SVM is however less familiar to most of us. +It relies on the definition of hyperplanes and the +definition of a margin which separates classes (in case of +classification problems) of variables. It is also used for regression +problems. +

    + +

    With SVMs we distinguish between hard margin and soft margins. The +latter introduces a so-called softening parameter to be discussed +below. We distinguish also between linear and non-linear +approaches. The latter are the most frequent ones since it is rather +unlikely that we can separate classes easily by say straight lines. +

    + +









    +

    Hyperplanes and all that

    + +

    The theory behind support vector machines (SVM hereafter) is based on +the mathematical description of so-called hyperplanes. Let us start +with a two-dimensional case. This will also allow us to introduce our +first SVM examples. These will be tailored to the case of two specific +classes, as displayed in the figure here based on the usage of the petal data. +

    + +

    We assume here that our data set can be well separated into two +domains, where a straight line does the job in the separating the two +classes. Here the two classes are represented by either squares or +circles. +

    + + +
    +
    +
    +
    +
    +
    from sklearn import datasets
    +from sklearn.svm import SVC, LinearSVC
    +from sklearn.linear_model import SGDClassifier
    +from sklearn.preprocessing import StandardScaler
    +import matplotlib
    +import matplotlib.pyplot as plt
    +plt.rcParams['axes.labelsize'] = 14
    +plt.rcParams['xtick.labelsize'] = 12
    +plt.rcParams['ytick.labelsize'] = 12
    +
    +
    +iris = datasets.load_iris()
    +X = iris["data"][:, (2, 3)]  # petal length, petal width
    +y = iris["target"]
    +
    +setosa_or_versicolor = (y == 0) | (y == 1)
    +X = X[setosa_or_versicolor]
    +y = y[setosa_or_versicolor]
    +
    +
    +
    +C = 5
    +alpha = 1 / (C * len(X))
    +
    +lin_clf = LinearSVC(loss="hinge", C=C, random_state=42)
    +svm_clf = SVC(kernel="linear", C=C)
    +sgd_clf = SGDClassifier(loss="hinge", learning_rate="constant", eta0=0.001, alpha=alpha,
    +                        max_iter=100000, random_state=42)
    +
    +scaler = StandardScaler()
    +X_scaled = scaler.fit_transform(X)
    +
    +lin_clf.fit(X_scaled, y)
    +svm_clf.fit(X_scaled, y)
    +sgd_clf.fit(X_scaled, y)
    +
    +print("LinearSVC:                   ", lin_clf.intercept_, lin_clf.coef_)
    +print("SVC:                         ", svm_clf.intercept_, svm_clf.coef_)
    +print("SGDClassifier(alpha={:.5f}):".format(sgd_clf.alpha), sgd_clf.intercept_, sgd_clf.coef_)
    +
    +# Compute the slope and bias of each decision boundary
    +w1 = -lin_clf.coef_[0, 0]/lin_clf.coef_[0, 1]
    +b1 = -lin_clf.intercept_[0]/lin_clf.coef_[0, 1]
    +w2 = -svm_clf.coef_[0, 0]/svm_clf.coef_[0, 1]
    +b2 = -svm_clf.intercept_[0]/svm_clf.coef_[0, 1]
    +w3 = -sgd_clf.coef_[0, 0]/sgd_clf.coef_[0, 1]
    +b3 = -sgd_clf.intercept_[0]/sgd_clf.coef_[0, 1]
    +
    +# Transform the decision boundary lines back to the original scale
    +line1 = scaler.inverse_transform([[-10, -10 * w1 + b1], [10, 10 * w1 + b1]])
    +line2 = scaler.inverse_transform([[-10, -10 * w2 + b2], [10, 10 * w2 + b2]])
    +line3 = scaler.inverse_transform([[-10, -10 * w3 + b3], [10, 10 * w3 + b3]])
    +
    +# Plot all three decision boundaries
    +plt.figure(figsize=(11, 4))
    +plt.plot(line1[:, 0], line1[:, 1], "k:", label="LinearSVC")
    +plt.plot(line2[:, 0], line2[:, 1], "b--", linewidth=2, label="SVC")
    +plt.plot(line3[:, 0], line3[:, 1], "r-", label="SGDClassifier")
    +plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs") # label="Iris-Versicolor"
    +plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo") # label="Iris-Setosa"
    +plt.xlabel("Petal length", fontsize=14)
    +plt.ylabel("Petal width", fontsize=14)
    +plt.legend(loc="upper center", fontsize=14)
    +plt.axis([0, 5.5, 0, 2])
    +
    +plt.show()
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + + +









    +

    What is a hyperplane?

    + +

    The aim of the SVM algorithm is to find a hyperplane in a +\( p \)-dimensional space, where \( p \) is the number of features that +distinctly classifies the data points. +

    + +

    In a \( p \)-dimensional space, a hyperplane is what we call an affine subspace of dimension of \( p-1 \). +As an example, in two dimension, a hyperplane is simply as straight line while in three dimensions it is +a two-dimensional subspace, or stated simply, a plane. +

    + +

    In two dimensions, with the variables \( x_1 \) and \( x_2 \), the hyperplane is defined as

    +$$ +b+w_1x_1+w_2x_2=0, +$$ + +

    where \( b \) is the intercept and \( w_1 \) and \( w_2 \) define the elements of a vector orthogonal to the line +\( b+w_1x_1+w_2x_2=0 \). +In two dimensions we define the vectors \( \boldsymbol{x} =[x1,x2] \) and \( \boldsymbol{w}=[w1,w2] \). +We can then rewrite the above equation as +

    + +$$ +\boldsymbol{x}^T\boldsymbol{w}+b=0. +$$ + + +









    +

    A \( p \)-dimensional space of features

    + +

    We limit ourselves to two classes of outputs \( y_i \) and assign these classes the values \( y_i = \pm 1 \). +In a \( p \)-dimensional space of say \( p \) features we have a hyperplane defines as +

    +$$ +b+wx_1+w_2x_2+\dots +w_px_p=0. +$$ + +

    If we define a +matrix \( \boldsymbol{X}=\left[\boldsymbol{x}_1,\boldsymbol{x}_2,\dots, \boldsymbol{x}_p\right] \) +of dimension \( n\times p \), where \( n \) represents the observations for each feature and each vector \( x_i \) is a column vector of the matrix \( \boldsymbol{X} \), +

    +$$ +\boldsymbol{x}_i = \begin{bmatrix} x_{i1} \\ x_{i2} \\ \dots \\ \dots \\ x_{ip} \end{bmatrix}. +$$ + +

    If the above condition is not met for a given vector \( \boldsymbol{x}_i \) we have

    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} >0, +$$ + +

    if our output \( y_i=1 \). +In this case we say that \( \boldsymbol{x}_i \) lies on one of the sides of the hyperplane and if +

    +$$ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} < 0, +$$ + +

    for the class of observations \( y_i=-1 \), +then \( \boldsymbol{x}_i \) lies on the other side. +

    + +

    Equivalently, for the two classes of observations we have

    +$$ +y_i\left(b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip}\right) > 0. +$$ + +

    When we try to separate hyperplanes, if it exists, we can use it to construct a natural classifier: a test observation is assigned a given class depending on which side of the hyperplane it is located.

    + + +

    The two-dimensional case

    + +

    Let us try to develop our intuition about SVMs by limiting ourselves to a two-dimensional +plane. To separate the two classes of data points, there are many +possible lines (hyperplanes if you prefer a more strict naming) +that could be chosen. Our objective is to find a +plane that has the maximum margin, i.e the maximum distance between +data points of both classes. Maximizing the margin distance provides +some reinforcement so that future data points can be classified with +more confidence. +

    + +

    What a linear classifier attempts to accomplish is to split the +feature space into two half spaces by placing a hyperplane between the +data points. This hyperplane will be our decision boundary. All +points on one side of the plane will belong to class one and all points +on the other side of the plane will belong to the second class two. +

    + +

    Unfortunately there are many ways in which we can place a hyperplane +to divide the data. Below is an example of two candidate hyperplanes +for our data sample. +

    + +









    +

    Getting into the details

    + +

    Let us define the function

    +$$ +f(x) = \boldsymbol{w}^T\boldsymbol{x}+b = 0, +$$ + +

    as the function that determines the line \( L \) that separates two classes (our two features), see the figure here.

    + +

    Any point defined by \( \boldsymbol{x}_i \) and \( \boldsymbol{x}_2 \) on the line \( L \) will satisfy \( \boldsymbol{w}^T(\boldsymbol{x}_1-\boldsymbol{x}_2)=0 \).

    + +

    The signed distance \( \delta \) from any point defined by a vector \( \boldsymbol{x} \) and a point \( \boldsymbol{x}_0 \) on the line \( L \) is then

    +$$ +\delta = \frac{1}{\vert\vert \boldsymbol{w}\vert\vert}(\boldsymbol{w}^T\boldsymbol{x}+b). +$$ + + +









    +

    First attempt at a minimization approach

    + +

    How do we find the parameter \( b \) and the vector \( \boldsymbol{w} \)? What we could +do is to define a cost function which now contains the set of all +misclassified points \( M \) and attempt to minimize this function +

    + +$$ +C(\boldsymbol{w},b) = -\sum_{i\in M} y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ + +

    We could now for example define all values \( y_i =1 \) as misclassified in case we have \( \boldsymbol{w}^T\boldsymbol{x}_i+b < 0 \) and the opposite if we have \( y_i=-1 \). Taking the derivatives gives us

    +$$ +\frac{\partial C}{\partial b} = -\sum_{i\in M} y_i, +$$ + +

    and

    +$$ +\frac{\partial C}{\partial \boldsymbol{w}} = -\sum_{i\in M} y_ix_i. +$$ + + +









    +

    Solving the equations

    + +

    We can now use the Newton-Raphson method or different variants of the gradient descent family (from plain gradient descent to various stochastic gradient descent approaches) to solve the equations

    +$$ +b \leftarrow b +\eta \frac{\partial C}{\partial b}, +$$ + +

    and

    +$$ +\boldsymbol{w} \leftarrow \boldsymbol{w} +\eta \frac{\partial C}{\partial \boldsymbol{w}}, +$$ + +

    where \( \eta \) is our by now well-known learning rate.

    + +









    +

    Code Example

    + +

    The equations we discussed above can be coded rather easily (the +framework is similar to what we developed for logistic +regression). We are going to set up a simple case with two classes only and we want to find a line which separates them the best possible way. +

    + + +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    +
    + + +









    +

    Problems with the Simpler Approach

    + +

    There are however problems with this approach, although it looks +pretty straightforward to implement. When running the above code, we see that we can easily end up with many diffeent lines which separate the two classes. +

    + +

    For small +gaps between the entries, we may also end up needing many iterations +before the solutions converge and if the data cannot be separated +properly into two distinct classes, we may not experience a converge +at all. +

    + +









    +

    A better approach

    + +

    A better approach is rather to try to define a large margin between +the two classes (if they are well separated from the beginning). +

    + +

    Thus, we wish to find a margin \( M \) with \( \boldsymbol{w} \) normalized to +\( \vert\vert \boldsymbol{w}\vert\vert =1 \) subject to the condition +

    + +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, p. +$$ + +

    All points are thus at a signed distance from the decision boundary defined by the line \( L \). The parameters \( b \) and \( w_1 \) and \( w_2 \) define this line.

    + +

    We seek thus the largest value \( M \) defined by

    +$$ +\frac{1}{\vert \vert \boldsymbol{w}\vert\vert}y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, n, +$$ + +

    or just

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq M\vert \vert \boldsymbol{w}\vert\vert \hspace{0.1cm}\forall i. +$$ + +

    If we scale the equation so that \( \vert \vert \boldsymbol{w}\vert\vert = 1/M \), we have to find the minimum of +\( \boldsymbol{w}^T\boldsymbol{w}=\vert \vert \boldsymbol{w}\vert\vert \) (the norm) subject to the condition +

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) \geq 1 \hspace{0.1cm}\forall i. +$$ + +

    We have thus defined our margin as the invers of the norm of +\( \boldsymbol{w} \). We want to minimize the norm in order to have a as large as +possible margin \( M \). Before we proceed, we need to remind ourselves +about Lagrangian multipliers. +

    + +









    +

    A quick Reminder on Lagrangian Multipliers

    + +

    Consider a function of three independent variables \( f(x,y,z) \) . For the function \( f \) to be an +extreme we have +

    +$$ +df=0. +$$ + +

    A necessary and sufficient condition is

    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ + +

    due to

    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz. +$$ + +

    In many problems the variables \( x,y,z \) are often subject to constraints (such as those above for the margin) +so that they are no longer all independent. It is possible at least in principle to use each +constraint to eliminate one variable +and to proceed with a new and smaller set of independent varables. +

    + +

    The use of so-called Lagrangian multipliers is an alternative technique when the elimination +of variables is incovenient or undesirable. Assume that we have an equation of constraint on +the variables \( x,y,z \) +

    +$$ +\phi(x,y,z) = 0, +$$ + +

    resulting in

    +$$ +d\phi = \frac{\partial \phi}{\partial x}dx+\frac{\partial \phi}{\partial y}dy+\frac{\partial \phi}{\partial z}dz =0. +$$ + +

    Now we cannot set anymore

    +$$ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +$$ + +

    if \( df=0 \) is wanted +because there are now only two independent variables! Assume \( x \) and \( y \) are the independent +variables. +Then \( dz \) is no longer arbitrary. +

    + +









    +

    Adding the Multiplier

    + +

    However, we can add to

    +$$ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz, +$$ + +

    a multiplum of \( d\phi \), viz. \( \lambda d\phi \), resulting in

    +$$ +df+\lambda d\phi = (\frac{\partial f}{\partial z}+\lambda +\frac{\partial \phi}{\partial x})dx+(\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y})dy+ +(\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z})dz =0. +$$ + +

    Our multiplier is chosen so that

    +$$ +\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z} =0. +$$ + +

    We need to remember that we took \( dx \) and \( dy \) to be arbitrary and thus we must have

    +$$ +\frac{\partial f}{\partial x}+\lambda\frac{\partial \phi}{\partial x} =0, +$$ + +

    and

    +$$ +\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y} =0. +$$ + +

    When all these equations are satisfied, \( df=0 \). We have four unknowns, \( x,y,z \) and +\( \lambda \). Actually we want only \( x,y,z \), \( \lambda \) needs not to be determined, +it is therefore often called +Lagrange's undetermined multiplier. +If we have a set of constraints \( \phi_k \) we have the equations +

    +$$ +\frac{\partial f}{\partial x_i}+\sum_k\lambda_k\frac{\partial \phi_k}{\partial x_i} =0. +$$ + + +









    +

    Setting up the Problem

    +

    In order to solve the above problem, we define the following Lagrangian function to be minimized

    +$$ +{\cal L}(\lambda,b,\boldsymbol{w})=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-1\right], +$$ + +

    where \( \lambda_i \) is a so-called Lagrange multiplier subject to the condition \( \lambda_i \geq 0 \).

    + +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ + +

    and

    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ + +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    subject to the constraints \( \lambda_i\geq 0 \) and \( \sum_i\lambda_iy_i=0 \). +We must in addition satisfy the Karush-Kuhn-Tucker (KKT) condition +

    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -1\right] \hspace{0.1cm}\forall i. +$$ + +
      +
    1. If \( \lambda_i > 0 \), then \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) and we say that \( x_i \) is on the boundary.
    2. +
    3. If \( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)> 1 \), we say \( x_i \) is not on the boundary and we set \( \lambda_i=0 \).
    4. +
    +

    When \( \lambda_i > 0 \), the vectors \( \boldsymbol{x}_i \) are called support vectors. They are the vectors closest to the line (or hyperplane) and define the margin \( M \).

    + +









    +

    The problem to solve

    + +

    We can rewrite

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    and its constraints in terms of a matrix-vector problem where we minimize w.r.t. \( \lambda \) the following problem

    +$$ +\frac{1}{2} \boldsymbol{\lambda}^T\begin{bmatrix} y_1y_1\boldsymbol{x}_1^T\boldsymbol{x}_1 & y_1y_2\boldsymbol{x}_1^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_1^T\boldsymbol{x}_n \\ +y_2y_1\boldsymbol{x}_2^T\boldsymbol{x}_1 & y_2y_2\boldsymbol{x}_2^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_2^T\boldsymbol{x}_n \\ +\dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots \\ +y_ny_1\boldsymbol{x}_n^T\boldsymbol{x}_1 & y_ny_2\boldsymbol{x}_n^T\boldsymbol{x}_2 & \dots & \dots & y_ny_n\boldsymbol{x}_n^T\boldsymbol{x}_n \\ +\end{bmatrix}\boldsymbol{\lambda}-\mathbb{1}\boldsymbol{\lambda}, +$$ + +

    subject to \( \boldsymbol{y}^T\boldsymbol{\lambda}=0 \). Here we defined the vectors \( \boldsymbol{\lambda} =[\lambda_1,\lambda_2,\dots,\lambda_n] \) and +\( \boldsymbol{y}=[y_1,y_2,\dots,y_n] \). +

    + +









    +

    The last steps

    + +

    Solving the above problem, yields the values of \( \lambda_i \). +To find the coefficients of your hyperplane we need simply to compute +

    +$$ +\boldsymbol{w}=\sum_{i} \lambda_iy_i\boldsymbol{x}_i. +$$ + +

    With our vector \( \boldsymbol{w} \) we can in turn find the value of the intercept \( b \) (here in two dimensions) via

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ + +

    resulting in

    +$$ +b = \frac{1}{y_i}-\boldsymbol{w}^T\boldsymbol{x}_i, +$$ + +

    or if we write it out in terms of the support vectors only, with \( N_s \) being their number, we have

    +$$ +b = \frac{1}{N_s}\sum_{j\in N_s}\left(y_j-\sum_{i=1}^n\lambda_iy_i\boldsymbol{x}_i^T\boldsymbol{x}_j\right). +$$ + +

    With our hyperplane coefficients we can use our classifier to assign any observation by simply using

    +$$ +y_i = \mathrm{sign}(\boldsymbol{w}^T\boldsymbol{x}_i+b). +$$ + +

    Below we discuss how to find the optimal values of \( \lambda_i \). Before we proceed however, we discuss now the so-called soft classifier.

    + +









    +

    A soft classifier

    + +

    Till now, the margin is strictly defined by the support vectors. This defines what is called a hard classifier, that is the margins are well defined.

    + +

    Suppose now that classes overlap in feature space, as shown in the +figure here. One way to deal with this problem before we define the +so-called kernel approach, is to allow a kind of slack in the sense +that we allow some points to be on the wrong side of the margin. +

    + +

    We introduce thus the so-called slack variables \( \boldsymbol{\xi} =[\xi_1,x_2,\dots,x_n] \) and +modify our previous equation +

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1, +$$ + +

    to

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i, +$$ + +

    with the requirement \( \xi_i\geq 0 \). The total violation is now \( \sum_i\xi \). +The value \( \xi_i \) in the constraint the last constraint corresponds to the amount by which the prediction +\( y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1 \) is on the wrong side of its margin. Hence by bounding the sum \( \sum_i \xi_i \), +we bound the total amount by which predictions fall on the wrong side of their margins. +

    + +

    Misclassifications occur when \( \xi_i > 1 \). Thus bounding the total sum by some value \( C \) bounds in turn the total number of +misclassifications. +

    + +









    +

    Soft optmization problem

    + +

    This has in turn the consequences that we change our optmization problem to finding the minimum of

    +$$ +{\cal L}=\frac{1}{2}\boldsymbol{w}^T\boldsymbol{w}-\sum_{i=1}^n\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)-(1-\xi_)\right]+C\sum_{i=1}^n\xi_i-\sum_{i=1}^n\gamma_i\xi_i, +$$ + +

    subject to

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1-\xi_i \hspace{0.1cm}\forall i, +$$ + +

    with the requirement \( \xi_i\geq 0 \).

    + +

    Taking the derivatives with respect to \( b \) and \( \boldsymbol{w} \) we obtain

    +$$ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +$$ + +

    and

    +$$ +\frac{\partial {\cal L}}{\partial \boldsymbol{w}} = 0 = \boldsymbol{w}-\sum_{i} \lambda_iy_i\boldsymbol{x}_i, +$$ + +

    and

    +$$ +\lambda_i = C-\gamma_i \hspace{0.1cm}\forall i. +$$ + +

    Inserting these constraints into the equation for \( {\cal L} \) we obtain the same equation as before

    +$$ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j, +$$ + +

    but now subject to the constraints \( \lambda_i\geq 0 \), \( \sum_i\lambda_iy_i=0 \) and \( 0\leq\lambda_i \leq C \). +We must in addition satisfy the Karush-Kuhn-Tucker condition which now reads +

    +$$ +\lambda_i\left[y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_)\right]=0 \hspace{0.1cm}\forall i, +$$ + +$$ +\gamma_i\xi_i = 0, +$$ + +

    and

    +$$ +y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b) -(1-\xi_) \geq 0 \hspace{0.1cm}\forall i. +$$ +
    © 1999-2022, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license diff --git a/doc/pub/week45/ipynb/ipynb-week45-src.tar.gz b/doc/pub/week45/ipynb/ipynb-week45-src.tar.gz index 90bc939ba..94c9aff61 100644 Binary files a/doc/pub/week45/ipynb/ipynb-week45-src.tar.gz and b/doc/pub/week45/ipynb/ipynb-week45-src.tar.gz differ diff --git a/doc/pub/week45/ipynb/week45.ipynb b/doc/pub/week45/ipynb/week45.ipynb index dda082139..634b576f3 100644 --- a/doc/pub/week45/ipynb/week45.ipynb +++ b/doc/pub/week45/ipynb/week45.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "21428307", + "id": "14457a9e", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "08d01224", + "id": "3c420796", "metadata": { "editable": true }, @@ -29,7 +29,7 @@ }, { "cell_type": "markdown", - "id": "c369db1a", + "id": "78a9bea0", "metadata": { "editable": true }, @@ -59,7 +59,7 @@ }, { "cell_type": "markdown", - "id": "5da9e5af", + "id": "6df15f4d", "metadata": { "editable": true }, @@ -70,7 +70,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "c6c5bcbc", + "id": "2b7789cc", "metadata": { "collapsed": false, "editable": true @@ -170,7 +170,7 @@ }, { "cell_type": "markdown", - "id": "9e20ea77", + "id": "0dc5853c", "metadata": { "editable": true }, @@ -190,7 +190,7 @@ }, { "cell_type": "markdown", - "id": "666db1a9", + "id": "d653fd00", "metadata": { "editable": true }, @@ -204,7 +204,7 @@ }, { "cell_type": "markdown", - "id": "e8b04953", + "id": "bc9947ac", "metadata": { "editable": true }, @@ -216,7 +216,7 @@ }, { "cell_type": "markdown", - "id": "d8ac7060", + "id": "a96b4319", "metadata": { "editable": true }, @@ -233,7 +233,7 @@ }, { "cell_type": "markdown", - "id": "b8e65d8f", + "id": "f24d7f46", "metadata": { "editable": true }, @@ -245,7 +245,7 @@ }, { "cell_type": "markdown", - "id": "c02eaf83", + "id": "b19ecdc7", "metadata": { "editable": true }, @@ -259,7 +259,7 @@ }, { "cell_type": "markdown", - "id": "8edc1125", + "id": "eda428b4", "metadata": { "editable": true }, @@ -271,7 +271,7 @@ }, { "cell_type": "markdown", - "id": "b12948d4", + "id": "72bc7bde", "metadata": { "editable": true }, @@ -284,7 +284,7 @@ }, { "cell_type": "markdown", - "id": "5e327556", + "id": "1e08222d", "metadata": { "editable": true }, @@ -296,7 +296,7 @@ }, { "cell_type": "markdown", - "id": "e78e752f", + "id": "08b2b1a5", "metadata": { "editable": true }, @@ -306,7 +306,7 @@ }, { "cell_type": "markdown", - "id": "7640a1b7", + "id": "b29ac8f7", "metadata": { "editable": true }, @@ -334,7 +334,7 @@ }, { "cell_type": "markdown", - "id": "8c7c46fa", + "id": "962aa964", "metadata": { "editable": true }, @@ -350,7 +350,7 @@ }, { "cell_type": "markdown", - "id": "8c850c6f", + "id": "5f7ad153", "metadata": { "editable": true }, @@ -362,7 +362,7 @@ }, { "cell_type": "markdown", - "id": "df996614", + "id": "91be4e38", "metadata": { "editable": true }, @@ -373,7 +373,7 @@ }, { "cell_type": "markdown", - "id": "18614499", + "id": "34b20671", "metadata": { "editable": true }, @@ -385,7 +385,7 @@ }, { "cell_type": "markdown", - "id": "a7ff37d1", + "id": "0b2c6a73", "metadata": { "editable": true }, @@ -395,7 +395,7 @@ }, { "cell_type": "markdown", - "id": "7ddcbb2b", + "id": "cffa3355", "metadata": { "editable": true }, @@ -407,7 +407,7 @@ }, { "cell_type": "markdown", - "id": "5eb3ce9e", + "id": "071d0ffe", "metadata": { "editable": true }, @@ -417,7 +417,7 @@ }, { "cell_type": "markdown", - "id": "2a0ef624", + "id": "a64a88fb", "metadata": { "editable": true }, @@ -429,7 +429,7 @@ }, { "cell_type": "markdown", - "id": "81583e4a", + "id": "0e2a4384", "metadata": { "editable": true }, @@ -439,7 +439,7 @@ }, { "cell_type": "markdown", - "id": "9166221d", + "id": "5a3bbbe3", "metadata": { "editable": true }, @@ -451,7 +451,7 @@ }, { "cell_type": "markdown", - "id": "8852e311", + "id": "64affb8b", "metadata": { "editable": true }, @@ -465,7 +465,7 @@ }, { "cell_type": "markdown", - "id": "210179cd", + "id": "cf165a63", "metadata": { "editable": true }, @@ -481,7 +481,7 @@ }, { "cell_type": "markdown", - "id": "ad8a6061", + "id": "f17bc8b2", "metadata": { "editable": true }, @@ -493,7 +493,7 @@ }, { "cell_type": "markdown", - "id": "7a22ffeb", + "id": "2abb5152", "metadata": { "editable": true }, @@ -509,7 +509,7 @@ }, { "cell_type": "markdown", - "id": "14a61057", + "id": "e9bdd274", "metadata": { "editable": true }, @@ -521,7 +521,7 @@ }, { "cell_type": "markdown", - "id": "a00e889a", + "id": "97841488", "metadata": { "editable": true }, @@ -531,7 +531,7 @@ }, { "cell_type": "markdown", - "id": "079d03de", + "id": "8053c61c", "metadata": { "editable": true }, @@ -543,7 +543,7 @@ }, { "cell_type": "markdown", - "id": "375ca97b", + "id": "2bd0661a", "metadata": { "editable": true }, @@ -555,7 +555,7 @@ }, { "cell_type": "markdown", - "id": "823e9234", + "id": "69236806", "metadata": { "editable": true }, @@ -567,7 +567,7 @@ }, { "cell_type": "markdown", - "id": "d312a276", + "id": "cd80669b", "metadata": { "editable": true }, @@ -578,7 +578,7 @@ }, { "cell_type": "markdown", - "id": "c89a2245", + "id": "66e939d6", "metadata": { "editable": true }, @@ -590,7 +590,7 @@ }, { "cell_type": "markdown", - "id": "691f9c3c", + "id": "0ba3314f", "metadata": { "editable": true }, @@ -601,7 +601,7 @@ }, { "cell_type": "markdown", - "id": "57ec3ee5", + "id": "86fd49cf", "metadata": { "editable": true }, @@ -613,7 +613,7 @@ }, { "cell_type": "markdown", - "id": "0bdbd6db", + "id": "c1e5d151", "metadata": { "editable": true }, @@ -623,7 +623,7 @@ }, { "cell_type": "markdown", - "id": "e4ecfe32", + "id": "845eac84", "metadata": { "editable": true }, @@ -635,7 +635,7 @@ }, { "cell_type": "markdown", - "id": "91b5376a", + "id": "677ca6f1", "metadata": { "editable": true }, @@ -647,7 +647,7 @@ }, { "cell_type": "markdown", - "id": "fb40a6b2", + "id": "f463eb92", "metadata": { "editable": true }, @@ -659,7 +659,7 @@ }, { "cell_type": "markdown", - "id": "37294e7c", + "id": "686471ea", "metadata": { "editable": true }, @@ -671,7 +671,7 @@ }, { "cell_type": "markdown", - "id": "09d9568b", + "id": "37af88d9", "metadata": { "editable": true }, @@ -681,7 +681,7 @@ }, { "cell_type": "markdown", - "id": "25abc0b4", + "id": "994c663f", "metadata": { "editable": true }, @@ -693,7 +693,7 @@ }, { "cell_type": "markdown", - "id": "624377d3", + "id": "57e36fb8", "metadata": { "editable": true }, @@ -703,7 +703,7 @@ }, { "cell_type": "markdown", - "id": "a4f7a03b", + "id": "3722b60f", "metadata": { "editable": true }, @@ -715,7 +715,7 @@ }, { "cell_type": "markdown", - "id": "9f937b6c", + "id": "c6da31c0", "metadata": { "editable": true }, @@ -725,7 +725,7 @@ }, { "cell_type": "markdown", - "id": "9a3ce197", + "id": "03219ea4", "metadata": { "editable": true }, @@ -737,7 +737,7 @@ }, { "cell_type": "markdown", - "id": "5d6cf526", + "id": "ada50b55", "metadata": { "editable": true }, @@ -747,7 +747,7 @@ }, { "cell_type": "markdown", - "id": "4e8b88df", + "id": "adfe735c", "metadata": { "editable": true }, @@ -759,7 +759,7 @@ }, { "cell_type": "markdown", - "id": "08e10dc0", + "id": "6ffa64aa", "metadata": { "editable": true }, @@ -769,7 +769,7 @@ }, { "cell_type": "markdown", - "id": "66691c4f", + "id": "9b70e93e", "metadata": { "editable": true }, @@ -781,7 +781,7 @@ }, { "cell_type": "markdown", - "id": "f03ebd02", + "id": "87059f3a", "metadata": { "editable": true }, @@ -801,7 +801,7 @@ }, { "cell_type": "markdown", - "id": "69df9e32", + "id": "d51941f3", "metadata": { "editable": true }, @@ -813,7 +813,7 @@ }, { "cell_type": "markdown", - "id": "c6e9bd72", + "id": "6b2bac62", "metadata": { "editable": true }, @@ -823,7 +823,7 @@ }, { "cell_type": "markdown", - "id": "63255ef0", + "id": "d65c7454", "metadata": { "editable": true }, @@ -839,7 +839,7 @@ }, { "cell_type": "markdown", - "id": "4c5d96b6", + "id": "ac7c9fde", "metadata": { "editable": true }, @@ -851,7 +851,7 @@ }, { "cell_type": "markdown", - "id": "376315b5", + "id": "dc1ba2e3", "metadata": { "editable": true }, @@ -879,7 +879,7 @@ }, { "cell_type": "markdown", - "id": "9b515061", + "id": "e2747ce9", "metadata": { "editable": true }, @@ -892,7 +892,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "e67c859f", + "id": "7d153a8f", "metadata": { "collapsed": false, "editable": true @@ -917,7 +917,7 @@ }, { "cell_type": "markdown", - "id": "85e88cb1", + "id": "b12395a3", "metadata": { "editable": true }, @@ -935,7 +935,7 @@ }, { "cell_type": "markdown", - "id": "359ade7b", + "id": "bc35d977", "metadata": { "editable": true }, @@ -948,7 +948,7 @@ }, { "cell_type": "markdown", - "id": "14a03966", + "id": "dffabe6b", "metadata": { "editable": true }, @@ -960,7 +960,7 @@ }, { "cell_type": "markdown", - "id": "f6748296", + "id": "e9ad9109", "metadata": { "editable": true }, @@ -970,7 +970,7 @@ }, { "cell_type": "markdown", - "id": "d5931573", + "id": "e2c7ae96", "metadata": { "editable": true }, @@ -982,7 +982,7 @@ }, { "cell_type": "markdown", - "id": "cb0312aa", + "id": "d545e37c", "metadata": { "editable": true }, @@ -992,7 +992,7 @@ }, { "cell_type": "markdown", - "id": "d56b1260", + "id": "3f5d3579", "metadata": { "editable": true }, @@ -1004,7 +1004,7 @@ }, { "cell_type": "markdown", - "id": "e661c2a9", + "id": "aef07014", "metadata": { "editable": true }, @@ -1017,7 +1017,7 @@ }, { "cell_type": "markdown", - "id": "f0ccb6bb", + "id": "8020bfd9", "metadata": { "editable": true }, @@ -1029,7 +1029,7 @@ }, { "cell_type": "markdown", - "id": "153d21f6", + "id": "8fee4b5e", "metadata": { "editable": true }, @@ -1041,7 +1041,7 @@ }, { "cell_type": "markdown", - "id": "65518e0b", + "id": "3161ddcf", "metadata": { "editable": true }, @@ -1053,7 +1053,7 @@ }, { "cell_type": "markdown", - "id": "c81a6bf4", + "id": "cb405752", "metadata": { "editable": true }, @@ -1063,7 +1063,7 @@ }, { "cell_type": "markdown", - "id": "e5bdda53", + "id": "723c02b2", "metadata": { "editable": true }, @@ -1075,7 +1075,7 @@ }, { "cell_type": "markdown", - "id": "9a0c941c", + "id": "91bb5b6c", "metadata": { "editable": true }, @@ -1085,7 +1085,7 @@ }, { "cell_type": "markdown", - "id": "d25e46e2", + "id": "f5dfaa65", "metadata": { "editable": true }, @@ -1101,7 +1101,7 @@ }, { "cell_type": "markdown", - "id": "9b2534be", + "id": "d85cf1b7", "metadata": { "editable": true }, @@ -1113,7 +1113,7 @@ }, { "cell_type": "markdown", - "id": "c3e69b96", + "id": "19ddf838", "metadata": { "editable": true }, @@ -1134,7 +1134,7 @@ }, { "cell_type": "markdown", - "id": "e3cae636", + "id": "38ec9a85", "metadata": { "editable": true }, @@ -1145,7 +1145,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "258d09ae", + "id": "9b49cdcb", "metadata": { "collapsed": false, "editable": true @@ -1197,7 +1197,7 @@ }, { "cell_type": "markdown", - "id": "89889dd7", + "id": "f8629a16", "metadata": { "editable": true }, @@ -1208,7 +1208,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "de23da37", + "id": "666b4608", "metadata": { "collapsed": false, "editable": true @@ -1259,7 +1259,7 @@ }, { "cell_type": "markdown", - "id": "4bb0195a", + "id": "14524beb", "metadata": { "editable": true }, @@ -1282,7 +1282,7 @@ }, { "cell_type": "markdown", - "id": "f5a94a7a", + "id": "a1d5317e", "metadata": { "editable": true }, @@ -1293,7 +1293,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "769a0fe3", + "id": "4261548c", "metadata": { "collapsed": false, "editable": true @@ -1345,7 +1345,7 @@ }, { "cell_type": "markdown", - "id": "0e1ef278", + "id": "936e99e7", "metadata": { "editable": true }, @@ -1358,7 +1358,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "f98213e3", + "id": "cbe49404", "metadata": { "collapsed": false, "editable": true @@ -1418,6 +1418,1544 @@ "save_fig(\"xgparams\")\n", "plt.show()" ] + }, + { + "cell_type": "markdown", + "id": "2663d377", + "metadata": { + "editable": true + }, + "source": [ + "## Support Vector Machines, overarching aims\n", + "\n", + "A Support Vector Machine (SVM) is a very powerful and versatile\n", + "Machine Learning method, capable of performing linear or nonlinear\n", + "classification, regression, and even outlier detection. It is one of\n", + "the most popular models in Machine Learning, and anyone interested in\n", + "Machine Learning should have it in their toolbox. SVMs are\n", + "particularly well suited for classification of complex but small-sized or\n", + "medium-sized datasets. \n", + "\n", + "The case with two well-separated classes only can be understood in an\n", + "intuitive way in terms of lines in a two-dimensional space separating\n", + "the two classes (see figure below).\n", + "\n", + "The basic mathematics behind the SVM is however less familiar to most of us. \n", + "It relies on the definition of hyperplanes and the\n", + "definition of a **margin** which separates classes (in case of\n", + "classification problems) of variables. It is also used for regression\n", + "problems.\n", + "\n", + "With SVMs we distinguish between hard margin and soft margins. The\n", + "latter introduces a so-called softening parameter to be discussed\n", + "below. We distinguish also between linear and non-linear\n", + "approaches. The latter are the most frequent ones since it is rather\n", + "unlikely that we can separate classes easily by say straight lines." + ] + }, + { + "cell_type": "markdown", + "id": "9b01af98", + "metadata": { + "editable": true + }, + "source": [ + "## Hyperplanes and all that\n", + "\n", + "The theory behind support vector machines (SVM hereafter) is based on\n", + "the mathematical description of so-called hyperplanes. Let us start\n", + "with a two-dimensional case. This will also allow us to introduce our\n", + "first SVM examples. These will be tailored to the case of two specific\n", + "classes, as displayed in the figure here based on the usage of the petal data.\n", + "\n", + "We assume here that our data set can be well separated into two\n", + "domains, where a straight line does the job in the separating the two\n", + "classes. Here the two classes are represented by either squares or\n", + "circles." + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "id": "c9523302", + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn import datasets\n", + "from sklearn.svm import SVC, LinearSVC\n", + "from sklearn.linear_model import SGDClassifier\n", + "from sklearn.preprocessing import StandardScaler\n", + "import matplotlib\n", + "import matplotlib.pyplot as plt\n", + "plt.rcParams['axes.labelsize'] = 14\n", + "plt.rcParams['xtick.labelsize'] = 12\n", + "plt.rcParams['ytick.labelsize'] = 12\n", + "\n", + "\n", + "iris = datasets.load_iris()\n", + "X = iris[\"data\"][:, (2, 3)] # petal length, petal width\n", + "y = iris[\"target\"]\n", + "\n", + "setosa_or_versicolor = (y == 0) | (y == 1)\n", + "X = X[setosa_or_versicolor]\n", + "y = y[setosa_or_versicolor]\n", + "\n", + "\n", + "\n", + "C = 5\n", + "alpha = 1 / (C * len(X))\n", + "\n", + "lin_clf = LinearSVC(loss=\"hinge\", C=C, random_state=42)\n", + "svm_clf = SVC(kernel=\"linear\", C=C)\n", + "sgd_clf = SGDClassifier(loss=\"hinge\", learning_rate=\"constant\", eta0=0.001, alpha=alpha,\n", + " max_iter=100000, random_state=42)\n", + "\n", + "scaler = StandardScaler()\n", + "X_scaled = scaler.fit_transform(X)\n", + "\n", + "lin_clf.fit(X_scaled, y)\n", + "svm_clf.fit(X_scaled, y)\n", + "sgd_clf.fit(X_scaled, y)\n", + "\n", + "print(\"LinearSVC: \", lin_clf.intercept_, lin_clf.coef_)\n", + "print(\"SVC: \", svm_clf.intercept_, svm_clf.coef_)\n", + "print(\"SGDClassifier(alpha={:.5f}):\".format(sgd_clf.alpha), sgd_clf.intercept_, sgd_clf.coef_)\n", + "\n", + "# Compute the slope and bias of each decision boundary\n", + "w1 = -lin_clf.coef_[0, 0]/lin_clf.coef_[0, 1]\n", + "b1 = -lin_clf.intercept_[0]/lin_clf.coef_[0, 1]\n", + "w2 = -svm_clf.coef_[0, 0]/svm_clf.coef_[0, 1]\n", + "b2 = -svm_clf.intercept_[0]/svm_clf.coef_[0, 1]\n", + "w3 = -sgd_clf.coef_[0, 0]/sgd_clf.coef_[0, 1]\n", + "b3 = -sgd_clf.intercept_[0]/sgd_clf.coef_[0, 1]\n", + "\n", + "# Transform the decision boundary lines back to the original scale\n", + "line1 = scaler.inverse_transform([[-10, -10 * w1 + b1], [10, 10 * w1 + b1]])\n", + "line2 = scaler.inverse_transform([[-10, -10 * w2 + b2], [10, 10 * w2 + b2]])\n", + "line3 = scaler.inverse_transform([[-10, -10 * w3 + b3], [10, 10 * w3 + b3]])\n", + "\n", + "# Plot all three decision boundaries\n", + "plt.figure(figsize=(11, 4))\n", + "plt.plot(line1[:, 0], line1[:, 1], \"k:\", label=\"LinearSVC\")\n", + "plt.plot(line2[:, 0], line2[:, 1], \"b--\", linewidth=2, label=\"SVC\")\n", + "plt.plot(line3[:, 0], line3[:, 1], \"r-\", label=\"SGDClassifier\")\n", + "plt.plot(X[:, 0][y==1], X[:, 1][y==1], \"bs\") # label=\"Iris-Versicolor\"\n", + "plt.plot(X[:, 0][y==0], X[:, 1][y==0], \"yo\") # label=\"Iris-Setosa\"\n", + "plt.xlabel(\"Petal length\", fontsize=14)\n", + "plt.ylabel(\"Petal width\", fontsize=14)\n", + "plt.legend(loc=\"upper center\", fontsize=14)\n", + "plt.axis([0, 5.5, 0, 2])\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "5d3e694d", + "metadata": { + "editable": true + }, + "source": [ + "## What is a hyperplane?\n", + "\n", + "The aim of the SVM algorithm is to find a hyperplane in a\n", + "$p$-dimensional space, where $p$ is the number of features that\n", + "distinctly classifies the data points.\n", + "\n", + "In a $p$-dimensional space, a hyperplane is what we call an affine subspace of dimension of $p-1$.\n", + "As an example, in two dimension, a hyperplane is simply as straight line while in three dimensions it is \n", + "a two-dimensional subspace, or stated simply, a plane. \n", + "\n", + "In two dimensions, with the variables $x_1$ and $x_2$, the hyperplane is defined as" + ] + }, + { + "cell_type": "markdown", + "id": "41879878", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b+w_1x_1+w_2x_2=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "8179cdc9", + "metadata": { + "editable": true + }, + "source": [ + "where $b$ is the intercept and $w_1$ and $w_2$ define the elements of a vector orthogonal to the line \n", + "$b+w_1x_1+w_2x_2=0$. \n", + "In two dimensions we define the vectors $\\boldsymbol{x} =[x1,x2]$ and $\\boldsymbol{w}=[w1,w2]$. \n", + "We can then rewrite the above equation as" + ] + }, + { + "cell_type": "markdown", + "id": "792425f1", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\boldsymbol{x}^T\\boldsymbol{w}+b=0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "ff43307d", + "metadata": { + "editable": true + }, + "source": [ + "## A $p$-dimensional space of features\n", + "\n", + "We limit ourselves to two classes of outputs $y_i$ and assign these classes the values $y_i = \\pm 1$. \n", + "In a $p$-dimensional space of say $p$ features we have a hyperplane defines as" + ] + }, + { + "cell_type": "markdown", + "id": "eab84086", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b+wx_1+w_2x_2+\\dots +w_px_p=0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "99ff7f1f", + "metadata": { + "editable": true + }, + "source": [ + "If we define a \n", + "matrix $\\boldsymbol{X}=\\left[\\boldsymbol{x}_1,\\boldsymbol{x}_2,\\dots, \\boldsymbol{x}_p\\right]$\n", + "of dimension $n\\times p$, where $n$ represents the observations for each feature and each vector $x_i$ is a column vector of the matrix $\\boldsymbol{X}$," + ] + }, + { + "cell_type": "markdown", + "id": "48409232", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\boldsymbol{x}_i = \\begin{bmatrix} x_{i1} \\\\ x_{i2} \\\\ \\dots \\\\ \\dots \\\\ x_{ip} \\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "2a893d96", + "metadata": { + "editable": true + }, + "source": [ + "If the above condition is not met for a given vector $\\boldsymbol{x}_i$ we have" + ] + }, + { + "cell_type": "markdown", + "id": "50cf85ef", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b+w_1x_{i1}+w_2x_{i2}+\\dots +w_px_{ip} >0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "c0d40a2e", + "metadata": { + "editable": true + }, + "source": [ + "if our output $y_i=1$.\n", + "In this case we say that $\\boldsymbol{x}_i$ lies on one of the sides of the hyperplane and if" + ] + }, + { + "cell_type": "markdown", + "id": "6025bd8b", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b+w_1x_{i1}+w_2x_{i2}+\\dots +w_px_{ip} < 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "55eb5f0e", + "metadata": { + "editable": true + }, + "source": [ + "for the class of observations $y_i=-1$, \n", + "then $\\boldsymbol{x}_i$ lies on the other side. \n", + "\n", + "Equivalently, for the two classes of observations we have" + ] + }, + { + "cell_type": "markdown", + "id": "c2974802", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i\\left(b+w_1x_{i1}+w_2x_{i2}+\\dots +w_px_{ip}\\right) > 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "c3cc8d3e", + "metadata": { + "editable": true + }, + "source": [ + "When we try to separate hyperplanes, if it exists, we can use it to construct a natural classifier: a test observation is assigned a given class depending on which side of the hyperplane it is located." + ] + }, + { + "cell_type": "markdown", + "id": "ecdd0c99", + "metadata": { + "editable": true + }, + "source": [ + "## The two-dimensional case\n", + "\n", + "Let us try to develop our intuition about SVMs by limiting ourselves to a two-dimensional\n", + "plane. To separate the two classes of data points, there are many\n", + "possible lines (hyperplanes if you prefer a more strict naming) \n", + "that could be chosen. Our objective is to find a\n", + "plane that has the maximum margin, i.e the maximum distance between\n", + "data points of both classes. Maximizing the margin distance provides\n", + "some reinforcement so that future data points can be classified with\n", + "more confidence.\n", + "\n", + "What a linear classifier attempts to accomplish is to split the\n", + "feature space into two half spaces by placing a hyperplane between the\n", + "data points. This hyperplane will be our decision boundary. All\n", + "points on one side of the plane will belong to class one and all points\n", + "on the other side of the plane will belong to the second class two.\n", + "\n", + "Unfortunately there are many ways in which we can place a hyperplane\n", + "to divide the data. Below is an example of two candidate hyperplanes\n", + "for our data sample." + ] + }, + { + "cell_type": "markdown", + "id": "1def394a", + "metadata": { + "editable": true + }, + "source": [ + "## Getting into the details\n", + "\n", + "Let us define the function" + ] + }, + { + "cell_type": "markdown", + "id": "2f2622d7", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "f(x) = \\boldsymbol{w}^T\\boldsymbol{x}+b = 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "5e1ffe21", + "metadata": { + "editable": true + }, + "source": [ + "as the function that determines the line $L$ that separates two classes (our two features), see the figure here. \n", + "\n", + "Any point defined by $\\boldsymbol{x}_i$ and $\\boldsymbol{x}_2$ on the line $L$ will satisfy $\\boldsymbol{w}^T(\\boldsymbol{x}_1-\\boldsymbol{x}_2)=0$. \n", + "\n", + "The signed distance $\\delta$ from any point defined by a vector $\\boldsymbol{x}$ and a point $\\boldsymbol{x}_0$ on the line $L$ is then" + ] + }, + { + "cell_type": "markdown", + "id": "96c1eb91", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\delta = \\frac{1}{\\vert\\vert \\boldsymbol{w}\\vert\\vert}(\\boldsymbol{w}^T\\boldsymbol{x}+b).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "f2ce6923", + "metadata": { + "editable": true + }, + "source": [ + "## First attempt at a minimization approach\n", + "\n", + "How do we find the parameter $b$ and the vector $\\boldsymbol{w}$? What we could\n", + "do is to define a cost function which now contains the set of all\n", + "misclassified points $M$ and attempt to minimize this function" + ] + }, + { + "cell_type": "markdown", + "id": "e4eb149c", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "C(\\boldsymbol{w},b) = -\\sum_{i\\in M} y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "193ce677", + "metadata": { + "editable": true + }, + "source": [ + "We could now for example define all values $y_i =1$ as misclassified in case we have $\\boldsymbol{w}^T\\boldsymbol{x}_i+b < 0$ and the opposite if we have $y_i=-1$. Taking the derivatives gives us" + ] + }, + { + "cell_type": "markdown", + "id": "b3598dd5", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial C}{\\partial b} = -\\sum_{i\\in M} y_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "02461076", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "e2692213", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial C}{\\partial \\boldsymbol{w}} = -\\sum_{i\\in M} y_ix_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "53c6c9b1", + "metadata": { + "editable": true + }, + "source": [ + "## Solving the equations\n", + "\n", + "We can now use the Newton-Raphson method or different variants of the gradient descent family (from plain gradient descent to various stochastic gradient descent approaches) to solve the equations" + ] + }, + { + "cell_type": "markdown", + "id": "14cfd4d5", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b \\leftarrow b +\\eta \\frac{\\partial C}{\\partial b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "07c8d24f", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "46ba9379", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\boldsymbol{w} \\leftarrow \\boldsymbol{w} +\\eta \\frac{\\partial C}{\\partial \\boldsymbol{w}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "479a8982", + "metadata": { + "editable": true + }, + "source": [ + "where $\\eta$ is our by now well-known learning rate." + ] + }, + { + "cell_type": "markdown", + "id": "50f225d0", + "metadata": { + "editable": true + }, + "source": [ + "## Code Example\n", + "\n", + "The equations we discussed above can be coded rather easily (the\n", + "framework is similar to what we developed for logistic\n", + "regression). We are going to set up a simple case with two classes only and we want to find a line which separates them the best possible way." + ] + }, + { + "cell_type": "markdown", + "id": "1d80aad8", + "metadata": { + "editable": true + }, + "source": [ + "## Problems with the Simpler Approach\n", + "\n", + "There are however problems with this approach, although it looks\n", + "pretty straightforward to implement. When running the above code, we see that we can easily end up with many diffeent lines which separate the two classes.\n", + "\n", + "For small\n", + "gaps between the entries, we may also end up needing many iterations\n", + "before the solutions converge and if the data cannot be separated\n", + "properly into two distinct classes, we may not experience a converge\n", + "at all." + ] + }, + { + "cell_type": "markdown", + "id": "d50aa9f6", + "metadata": { + "editable": true + }, + "source": [ + "## A better approach\n", + "\n", + "A better approach is rather to try to define a large margin between\n", + "the two classes (if they are well separated from the beginning).\n", + "\n", + "Thus, we wish to find a margin $M$ with $\\boldsymbol{w}$ normalized to\n", + "$\\vert\\vert \\boldsymbol{w}\\vert\\vert =1$ subject to the condition" + ] + }, + { + "cell_type": "markdown", + "id": "c0fb9cc9", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) \\geq M \\hspace{0.1cm}\\forall i=1,2,\\dots, p.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "42451f0a", + "metadata": { + "editable": true + }, + "source": [ + "All points are thus at a signed distance from the decision boundary defined by the line $L$. The parameters $b$ and $w_1$ and $w_2$ define this line. \n", + "\n", + "We seek thus the largest value $M$ defined by" + ] + }, + { + "cell_type": "markdown", + "id": "5fe03cb0", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{1}{\\vert \\vert \\boldsymbol{w}\\vert\\vert}y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) \\geq M \\hspace{0.1cm}\\forall i=1,2,\\dots, n,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "73d9ca1d", + "metadata": { + "editable": true + }, + "source": [ + "or just" + ] + }, + { + "cell_type": "markdown", + "id": "04b0d3f4", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) \\geq M\\vert \\vert \\boldsymbol{w}\\vert\\vert \\hspace{0.1cm}\\forall i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "3de2cd6c", + "metadata": { + "editable": true + }, + "source": [ + "If we scale the equation so that $\\vert \\vert \\boldsymbol{w}\\vert\\vert = 1/M$, we have to find the minimum of \n", + "$\\boldsymbol{w}^T\\boldsymbol{w}=\\vert \\vert \\boldsymbol{w}\\vert\\vert$ (the norm) subject to the condition" + ] + }, + { + "cell_type": "markdown", + "id": "4953f2d9", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) \\geq 1 \\hspace{0.1cm}\\forall i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "70954111", + "metadata": { + "editable": true + }, + "source": [ + "We have thus defined our margin as the invers of the norm of\n", + "$\\boldsymbol{w}$. We want to minimize the norm in order to have a as large as\n", + "possible margin $M$. Before we proceed, we need to remind ourselves\n", + "about Lagrangian multipliers." + ] + }, + { + "cell_type": "markdown", + "id": "0960d7b8", + "metadata": { + "editable": true + }, + "source": [ + "## A quick Reminder on Lagrangian Multipliers\n", + "\n", + "Consider a function of three independent variables $f(x,y,z)$ . For the function $f$ to be an\n", + "extreme we have" + ] + }, + { + "cell_type": "markdown", + "id": "2914bc19", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "df=0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "885c679a", + "metadata": { + "editable": true + }, + "source": [ + "A necessary and sufficient condition is" + ] + }, + { + "cell_type": "markdown", + "id": "2413e7c5", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial f}{\\partial x} =\\frac{\\partial f}{\\partial y}=\\frac{\\partial f}{\\partial z}=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "9926b0a0", + "metadata": { + "editable": true + }, + "source": [ + "due to" + ] + }, + { + "cell_type": "markdown", + "id": "4d16b4a7", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "df = \\frac{\\partial f}{\\partial x}dx+\\frac{\\partial f}{\\partial y}dy+\\frac{\\partial f}{\\partial z}dz.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "96f84dbb", + "metadata": { + "editable": true + }, + "source": [ + "In many problems the variables $x,y,z$ are often subject to constraints (such as those above for the margin)\n", + "so that they are no longer all independent. It is possible at least in principle to use each \n", + "constraint to eliminate one variable\n", + "and to proceed with a new and smaller set of independent varables.\n", + "\n", + "The use of so-called Lagrangian multipliers is an alternative technique when the elimination\n", + "of variables is incovenient or undesirable. Assume that we have an equation of constraint on \n", + "the variables $x,y,z$" + ] + }, + { + "cell_type": "markdown", + "id": "da5a4362", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\phi(x,y,z) = 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "ded8500c", + "metadata": { + "editable": true + }, + "source": [ + "resulting in" + ] + }, + { + "cell_type": "markdown", + "id": "1dfd6a92", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "d\\phi = \\frac{\\partial \\phi}{\\partial x}dx+\\frac{\\partial \\phi}{\\partial y}dy+\\frac{\\partial \\phi}{\\partial z}dz =0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "9c317ffe", + "metadata": { + "editable": true + }, + "source": [ + "Now we cannot set anymore" + ] + }, + { + "cell_type": "markdown", + "id": "b764c253", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial f}{\\partial x} =\\frac{\\partial f}{\\partial y}=\\frac{\\partial f}{\\partial z}=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "9701ec02", + "metadata": { + "editable": true + }, + "source": [ + "if $df=0$ is wanted\n", + "because there are now only two independent variables! Assume $x$ and $y$ are the independent \n", + "variables.\n", + "Then $dz$ is no longer arbitrary." + ] + }, + { + "cell_type": "markdown", + "id": "e8efa5de", + "metadata": { + "editable": true + }, + "source": [ + "## Adding the Multiplier\n", + "\n", + "However, we can add to" + ] + }, + { + "cell_type": "markdown", + "id": "ed746e0e", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "df = \\frac{\\partial f}{\\partial x}dx+\\frac{\\partial f}{\\partial y}dy+\\frac{\\partial f}{\\partial z}dz,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0a0eb329", + "metadata": { + "editable": true + }, + "source": [ + "a multiplum of $d\\phi$, viz. $\\lambda d\\phi$, resulting in" + ] + }, + { + "cell_type": "markdown", + "id": "d6c4f56d", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "df+\\lambda d\\phi = (\\frac{\\partial f}{\\partial z}+\\lambda\n", + "\\frac{\\partial \\phi}{\\partial x})dx+(\\frac{\\partial f}{\\partial y}+\\lambda\\frac{\\partial \\phi}{\\partial y})dy+\n", + "(\\frac{\\partial f}{\\partial z}+\\lambda\\frac{\\partial \\phi}{\\partial z})dz =0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "826fef05", + "metadata": { + "editable": true + }, + "source": [ + "Our multiplier is chosen so that" + ] + }, + { + "cell_type": "markdown", + "id": "f58c88b7", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial f}{\\partial z}+\\lambda\\frac{\\partial \\phi}{\\partial z} =0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "aae2d9cc", + "metadata": { + "editable": true + }, + "source": [ + "We need to remember that we took $dx$ and $dy$ to be arbitrary and thus we must have" + ] + }, + { + "cell_type": "markdown", + "id": "c4e3acbf", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial f}{\\partial x}+\\lambda\\frac{\\partial \\phi}{\\partial x} =0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0a8e0f61", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "988a0d97", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial f}{\\partial y}+\\lambda\\frac{\\partial \\phi}{\\partial y} =0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "17626433", + "metadata": { + "editable": true + }, + "source": [ + "When all these equations are satisfied, $df=0$. We have four unknowns, $x,y,z$ and\n", + "$\\lambda$. Actually we want only $x,y,z$, $\\lambda$ needs not to be determined, \n", + "it is therefore often called\n", + "Lagrange's undetermined multiplier.\n", + "If we have a set of constraints $\\phi_k$ we have the equations" + ] + }, + { + "cell_type": "markdown", + "id": "80944e18", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial f}{\\partial x_i}+\\sum_k\\lambda_k\\frac{\\partial \\phi_k}{\\partial x_i} =0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "bd982cc4", + "metadata": { + "editable": true + }, + "source": [ + "## Setting up the Problem\n", + "In order to solve the above problem, we define the following Lagrangian function to be minimized" + ] + }, + { + "cell_type": "markdown", + "id": "263c4bc0", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "{\\cal L}(\\lambda,b,\\boldsymbol{w})=\\frac{1}{2}\\boldsymbol{w}^T\\boldsymbol{w}-\\sum_{i=1}^n\\lambda_i\\left[y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)-1\\right],\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "fd3286d3", + "metadata": { + "editable": true + }, + "source": [ + "where $\\lambda_i$ is a so-called Lagrange multiplier subject to the condition $\\lambda_i \\geq 0$.\n", + "\n", + "Taking the derivatives with respect to $b$ and $\\boldsymbol{w}$ we obtain" + ] + }, + { + "cell_type": "markdown", + "id": "5c931bf6", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial {\\cal L}}{\\partial b} = -\\sum_{i} \\lambda_iy_i=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "176f4298", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "fedd0b98", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial {\\cal L}}{\\partial \\boldsymbol{w}} = 0 = \\boldsymbol{w}-\\sum_{i} \\lambda_iy_i\\boldsymbol{x}_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "6db24d25", + "metadata": { + "editable": true + }, + "source": [ + "Inserting these constraints into the equation for ${\\cal L}$ we obtain" + ] + }, + { + "cell_type": "markdown", + "id": "f2931d0e", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "{\\cal L}=\\sum_i\\lambda_i-\\frac{1}{2}\\sum_{ij}^n\\lambda_i\\lambda_jy_iy_j\\boldsymbol{x}_i^T\\boldsymbol{x}_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "6b76e747", + "metadata": { + "editable": true + }, + "source": [ + "subject to the constraints $\\lambda_i\\geq 0$ and $\\sum_i\\lambda_iy_i=0$. \n", + "We must in addition satisfy the [Karush-Kuhn-Tucker](https://en.wikipedia.org/wiki/Karush%E2%80%93Kuhn%E2%80%93Tucker_conditions) (KKT) condition" + ] + }, + { + "cell_type": "markdown", + "id": "edd12abe", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda_i\\left[y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) -1\\right] \\hspace{0.1cm}\\forall i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "7a7bc1c2", + "metadata": { + "editable": true + }, + "source": [ + "1. If $\\lambda_i > 0$, then $y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)=1$ and we say that $x_i$ is on the boundary.\n", + "\n", + "2. If $y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)> 1$, we say $x_i$ is not on the boundary and we set $\\lambda_i=0$. \n", + "\n", + "When $\\lambda_i > 0$, the vectors $\\boldsymbol{x}_i$ are called support vectors. They are the vectors closest to the line (or hyperplane) and define the margin $M$." + ] + }, + { + "cell_type": "markdown", + "id": "d063f2c9", + "metadata": { + "editable": true + }, + "source": [ + "## The problem to solve\n", + "\n", + "We can rewrite" + ] + }, + { + "cell_type": "markdown", + "id": "fc5b1886", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "{\\cal L}=\\sum_i\\lambda_i-\\frac{1}{2}\\sum_{ij}^n\\lambda_i\\lambda_jy_iy_j\\boldsymbol{x}_i^T\\boldsymbol{x}_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "6aa360c9", + "metadata": { + "editable": true + }, + "source": [ + "and its constraints in terms of a matrix-vector problem where we minimize w.r.t. $\\lambda$ the following problem" + ] + }, + { + "cell_type": "markdown", + "id": "5bb43661", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{1}{2} \\boldsymbol{\\lambda}^T\\begin{bmatrix} y_1y_1\\boldsymbol{x}_1^T\\boldsymbol{x}_1 & y_1y_2\\boldsymbol{x}_1^T\\boldsymbol{x}_2 & \\dots & \\dots & y_1y_n\\boldsymbol{x}_1^T\\boldsymbol{x}_n \\\\\n", + "y_2y_1\\boldsymbol{x}_2^T\\boldsymbol{x}_1 & y_2y_2\\boldsymbol{x}_2^T\\boldsymbol{x}_2 & \\dots & \\dots & y_1y_n\\boldsymbol{x}_2^T\\boldsymbol{x}_n \\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "y_ny_1\\boldsymbol{x}_n^T\\boldsymbol{x}_1 & y_ny_2\\boldsymbol{x}_n^T\\boldsymbol{x}_2 & \\dots & \\dots & y_ny_n\\boldsymbol{x}_n^T\\boldsymbol{x}_n \\\\\n", + "\\end{bmatrix}\\boldsymbol{\\lambda}-\\mathbb{1}\\boldsymbol{\\lambda},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "4cd04a25", + "metadata": { + "editable": true + }, + "source": [ + "subject to $\\boldsymbol{y}^T\\boldsymbol{\\lambda}=0$. Here we defined the vectors $\\boldsymbol{\\lambda} =[\\lambda_1,\\lambda_2,\\dots,\\lambda_n]$ and \n", + "$\\boldsymbol{y}=[y_1,y_2,\\dots,y_n]$." + ] + }, + { + "cell_type": "markdown", + "id": "db7bd4dd", + "metadata": { + "editable": true + }, + "source": [ + "## The last steps\n", + "\n", + "Solving the above problem, yields the values of $\\lambda_i$.\n", + "To find the coefficients of your hyperplane we need simply to compute" + ] + }, + { + "cell_type": "markdown", + "id": "3bfcc926", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\boldsymbol{w}=\\sum_{i} \\lambda_iy_i\\boldsymbol{x}_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "7bf59997", + "metadata": { + "editable": true + }, + "source": [ + "With our vector $\\boldsymbol{w}$ we can in turn find the value of the intercept $b$ (here in two dimensions) via" + ] + }, + { + "cell_type": "markdown", + "id": "939e1253", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)=1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "42920be4", + "metadata": { + "editable": true + }, + "source": [ + "resulting in" + ] + }, + { + "cell_type": "markdown", + "id": "fc34b422", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b = \\frac{1}{y_i}-\\boldsymbol{w}^T\\boldsymbol{x}_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "0b95735b", + "metadata": { + "editable": true + }, + "source": [ + "or if we write it out in terms of the support vectors only, with $N_s$ being their number, we have" + ] + }, + { + "cell_type": "markdown", + "id": "80576160", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "b = \\frac{1}{N_s}\\sum_{j\\in N_s}\\left(y_j-\\sum_{i=1}^n\\lambda_iy_i\\boldsymbol{x}_i^T\\boldsymbol{x}_j\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "aca2c413", + "metadata": { + "editable": true + }, + "source": [ + "With our hyperplane coefficients we can use our classifier to assign any observation by simply using" + ] + }, + { + "cell_type": "markdown", + "id": "9acf148b", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i = \\mathrm{sign}(\\boldsymbol{w}^T\\boldsymbol{x}_i+b).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "e8a5ceb3", + "metadata": { + "editable": true + }, + "source": [ + "Below we discuss how to find the optimal values of $\\lambda_i$. Before we proceed however, we discuss now the so-called soft classifier." + ] + }, + { + "cell_type": "markdown", + "id": "6f32c59c", + "metadata": { + "editable": true + }, + "source": [ + "## A soft classifier\n", + "\n", + "Till now, the margin is strictly defined by the support vectors. This defines what is called a hard classifier, that is the margins are well defined.\n", + "\n", + "Suppose now that classes overlap in feature space, as shown in the\n", + "figure here. One way to deal with this problem before we define the\n", + "so-called **kernel approach**, is to allow a kind of slack in the sense\n", + "that we allow some points to be on the wrong side of the margin.\n", + "\n", + "We introduce thus the so-called **slack** variables $\\boldsymbol{\\xi} =[\\xi_1,x_2,\\dots,x_n]$ and \n", + "modify our previous equation" + ] + }, + { + "cell_type": "markdown", + "id": "192e16cc", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)=1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "9152a401", + "metadata": { + "editable": true + }, + "source": [ + "to" + ] + }, + { + "cell_type": "markdown", + "id": "4bc13a1d", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)=1-\\xi_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "10cb3e42", + "metadata": { + "editable": true + }, + "source": [ + "with the requirement $\\xi_i\\geq 0$. The total violation is now $\\sum_i\\xi$. \n", + "The value $\\xi_i$ in the constraint the last constraint corresponds to the amount by which the prediction\n", + "$y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)=1$ is on the wrong side of its margin. Hence by bounding the sum $\\sum_i \\xi_i$,\n", + "we bound the total amount by which predictions fall on the wrong side of their margins.\n", + "\n", + "Misclassifications occur when $\\xi_i > 1$. Thus bounding the total sum by some value $C$ bounds in turn the total number of\n", + "misclassifications." + ] + }, + { + "cell_type": "markdown", + "id": "791dbc10", + "metadata": { + "editable": true + }, + "source": [ + "## Soft optmization problem\n", + "\n", + "This has in turn the consequences that we change our optmization problem to finding the minimum of" + ] + }, + { + "cell_type": "markdown", + "id": "2d420dc7", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "{\\cal L}=\\frac{1}{2}\\boldsymbol{w}^T\\boldsymbol{w}-\\sum_{i=1}^n\\lambda_i\\left[y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)-(1-\\xi_)\\right]+C\\sum_{i=1}^n\\xi_i-\\sum_{i=1}^n\\gamma_i\\xi_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "c25219b7", + "metadata": { + "editable": true + }, + "source": [ + "subject to" + ] + }, + { + "cell_type": "markdown", + "id": "ad4fb692", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b)=1-\\xi_i \\hspace{0.1cm}\\forall i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "a4b6eea3", + "metadata": { + "editable": true + }, + "source": [ + "with the requirement $\\xi_i\\geq 0$.\n", + "\n", + "Taking the derivatives with respect to $b$ and $\\boldsymbol{w}$ we obtain" + ] + }, + { + "cell_type": "markdown", + "id": "b433bc1b", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial {\\cal L}}{\\partial b} = -\\sum_{i} \\lambda_iy_i=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "05481a9d", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "800b5d20", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\frac{\\partial {\\cal L}}{\\partial \\boldsymbol{w}} = 0 = \\boldsymbol{w}-\\sum_{i} \\lambda_iy_i\\boldsymbol{x}_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "9db3c08a", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "483c6ed9", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda_i = C-\\gamma_i \\hspace{0.1cm}\\forall i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "d72d5a70", + "metadata": { + "editable": true + }, + "source": [ + "Inserting these constraints into the equation for ${\\cal L}$ we obtain the same equation as before" + ] + }, + { + "cell_type": "markdown", + "id": "bcc7c229", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "{\\cal L}=\\sum_i\\lambda_i-\\frac{1}{2}\\sum_{ij}^n\\lambda_i\\lambda_jy_iy_j\\boldsymbol{x}_i^T\\boldsymbol{x}_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "1046f34b", + "metadata": { + "editable": true + }, + "source": [ + "but now subject to the constraints $\\lambda_i\\geq 0$, $\\sum_i\\lambda_iy_i=0$ and $0\\leq\\lambda_i \\leq C$. \n", + "We must in addition satisfy the Karush-Kuhn-Tucker condition which now reads" + ] + }, + { + "cell_type": "markdown", + "id": "06444397", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\lambda_i\\left[y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) -(1-\\xi_)\\right]=0 \\hspace{0.1cm}\\forall i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "da502ea0", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "\\gamma_i\\xi_i = 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "id": "ea8de04e", + "metadata": { + "editable": true + }, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "id": "dd0bc989", + "metadata": { + "editable": true + }, + "source": [ + "$$\n", + "y_i(\\boldsymbol{w}^T\\boldsymbol{x}_i+b) -(1-\\xi_) \\geq 0 \\hspace{0.1cm}\\forall i.\n", + "$$" + ] } ], "metadata": {}, diff --git a/doc/src/week45/week45.do.txt b/doc/src/week45/week45.do.txt index 89d73dde3..b00057f44 100644 --- a/doc/src/week45/week45.do.txt +++ b/doc/src/week45/week45.do.txt @@ -744,8 +744,632 @@ plt.show() +!split +===== Support Vector Machines, overarching aims ===== + +A Support Vector Machine (SVM) is a very powerful and versatile +Machine Learning method, capable of performing linear or nonlinear +classification, regression, and even outlier detection. It is one of +the most popular models in Machine Learning, and anyone interested in +Machine Learning should have it in their toolbox. SVMs are +particularly well suited for classification of complex but small-sized or +medium-sized datasets. + +The case with two well-separated classes only can be understood in an +intuitive way in terms of lines in a two-dimensional space separating +the two classes (see figure below). + +The basic mathematics behind the SVM is however less familiar to most of us. +It relies on the definition of hyperplanes and the +definition of a _margin_ which separates classes (in case of +classification problems) of variables. It is also used for regression +problems. + +With SVMs we distinguish between hard margin and soft margins. The +latter introduces a so-called softening parameter to be discussed +below. We distinguish also between linear and non-linear +approaches. The latter are the most frequent ones since it is rather +unlikely that we can separate classes easily by say straight lines. + +!split +===== Hyperplanes and all that ===== + +The theory behind support vector machines (SVM hereafter) is based on +the mathematical description of so-called hyperplanes. Let us start +with a two-dimensional case. This will also allow us to introduce our +first SVM examples. These will be tailored to the case of two specific +classes, as displayed in the figure here based on the usage of the petal data. + +We assume here that our data set can be well separated into two +domains, where a straight line does the job in the separating the two +classes. Here the two classes are represented by either squares or +circles. +!bc pycod +from sklearn import datasets +from sklearn.svm import SVC, LinearSVC +from sklearn.linear_model import SGDClassifier +from sklearn.preprocessing import StandardScaler +import matplotlib +import matplotlib.pyplot as plt +plt.rcParams['axes.labelsize'] = 14 +plt.rcParams['xtick.labelsize'] = 12 +plt.rcParams['ytick.labelsize'] = 12 + + +iris = datasets.load_iris() +X = iris["data"][:, (2, 3)] # petal length, petal width +y = iris["target"] + +setosa_or_versicolor = (y == 0) | (y == 1) +X = X[setosa_or_versicolor] +y = y[setosa_or_versicolor] + + + +C = 5 +alpha = 1 / (C * len(X)) + +lin_clf = LinearSVC(loss="hinge", C=C, random_state=42) +svm_clf = SVC(kernel="linear", C=C) +sgd_clf = SGDClassifier(loss="hinge", learning_rate="constant", eta0=0.001, alpha=alpha, + max_iter=100000, random_state=42) + +scaler = StandardScaler() +X_scaled = scaler.fit_transform(X) + +lin_clf.fit(X_scaled, y) +svm_clf.fit(X_scaled, y) +sgd_clf.fit(X_scaled, y) + +print("LinearSVC: ", lin_clf.intercept_, lin_clf.coef_) +print("SVC: ", svm_clf.intercept_, svm_clf.coef_) +print("SGDClassifier(alpha={:.5f}):".format(sgd_clf.alpha), sgd_clf.intercept_, sgd_clf.coef_) + +# Compute the slope and bias of each decision boundary +w1 = -lin_clf.coef_[0, 0]/lin_clf.coef_[0, 1] +b1 = -lin_clf.intercept_[0]/lin_clf.coef_[0, 1] +w2 = -svm_clf.coef_[0, 0]/svm_clf.coef_[0, 1] +b2 = -svm_clf.intercept_[0]/svm_clf.coef_[0, 1] +w3 = -sgd_clf.coef_[0, 0]/sgd_clf.coef_[0, 1] +b3 = -sgd_clf.intercept_[0]/sgd_clf.coef_[0, 1] + +# Transform the decision boundary lines back to the original scale +line1 = scaler.inverse_transform([[-10, -10 * w1 + b1], [10, 10 * w1 + b1]]) +line2 = scaler.inverse_transform([[-10, -10 * w2 + b2], [10, 10 * w2 + b2]]) +line3 = scaler.inverse_transform([[-10, -10 * w3 + b3], [10, 10 * w3 + b3]]) + +# Plot all three decision boundaries +plt.figure(figsize=(11, 4)) +plt.plot(line1[:, 0], line1[:, 1], "k:", label="LinearSVC") +plt.plot(line2[:, 0], line2[:, 1], "b--", linewidth=2, label="SVC") +plt.plot(line3[:, 0], line3[:, 1], "r-", label="SGDClassifier") +plt.plot(X[:, 0][y==1], X[:, 1][y==1], "bs") # label="Iris-Versicolor" +plt.plot(X[:, 0][y==0], X[:, 1][y==0], "yo") # label="Iris-Setosa" +plt.xlabel("Petal length", fontsize=14) +plt.ylabel("Petal width", fontsize=14) +plt.legend(loc="upper center", fontsize=14) +plt.axis([0, 5.5, 0, 2]) + +plt.show() + + + +!ec +!split +===== What is a hyperplane? ===== + +The aim of the SVM algorithm is to find a hyperplane in a +$p$-dimensional space, where $p$ is the number of features that +distinctly classifies the data points. + +In a $p$-dimensional space, a hyperplane is what we call an affine subspace of dimension of $p-1$. +As an example, in two dimension, a hyperplane is simply as straight line while in three dimensions it is +a two-dimensional subspace, or stated simply, a plane. + +In two dimensions, with the variables $x_1$ and $x_2$, the hyperplane is defined as +!bt +\[ +b+w_1x_1+w_2x_2=0, +\] +!et + +where $b$ is the intercept and $w_1$ and $w_2$ define the elements of a vector orthogonal to the line +$b+w_1x_1+w_2x_2=0$. +In two dimensions we define the vectors $\bm{x} =[x1,x2]$ and $\bm{w}=[w1,w2]$. +We can then rewrite the above equation as + +!bt +\[ +\bm{x}^T\bm{w}+b=0. +\] +!et + +!split +===== A $p$-dimensional space of features ===== + +We limit ourselves to two classes of outputs $y_i$ and assign these classes the values $y_i = \pm 1$. +In a $p$-dimensional space of say $p$ features we have a hyperplane defines as +!bt +\[ +b+wx_1+w_2x_2+\dots +w_px_p=0. +\] +!et +If we define a +matrix $\bm{X}=\left[\bm{x}_1,\bm{x}_2,\dots, \bm{x}_p\right]$ +of dimension $n\times p$, where $n$ represents the observations for each feature and each vector $x_i$ is a column vector of the matrix $\bm{X}$, +!bt +\[ +\bm{x}_i = \begin{bmatrix} x_{i1} \\ x_{i2} \\ \dots \\ \dots \\ x_{ip} \end{bmatrix}. +\] +!et +If the above condition is not met for a given vector $\bm{x}_i$ we have +!bt +\[ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} >0, +\] +!et +if our output $y_i=1$. +In this case we say that $\bm{x}_i$ lies on one of the sides of the hyperplane and if +!bt +\[ +b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip} < 0, +\] +!et +for the class of observations $y_i=-1$, +then $\bm{x}_i$ lies on the other side. + +Equivalently, for the two classes of observations we have +!bt +\[ +y_i\left(b+w_1x_{i1}+w_2x_{i2}+\dots +w_px_{ip}\right) > 0. +\] +!et + +When we try to separate hyperplanes, if it exists, we can use it to construct a natural classifier: a test observation is assigned a given class depending on which side of the hyperplane it is located. + +!split +===== The two-dimensional case ===== + +Let us try to develop our intuition about SVMs by limiting ourselves to a two-dimensional +plane. To separate the two classes of data points, there are many +possible lines (hyperplanes if you prefer a more strict naming) +that could be chosen. Our objective is to find a +plane that has the maximum margin, i.e the maximum distance between +data points of both classes. Maximizing the margin distance provides +some reinforcement so that future data points can be classified with +more confidence. + +What a linear classifier attempts to accomplish is to split the +feature space into two half spaces by placing a hyperplane between the +data points. This hyperplane will be our decision boundary. All +points on one side of the plane will belong to class one and all points +on the other side of the plane will belong to the second class two. + +Unfortunately there are many ways in which we can place a hyperplane +to divide the data. Below is an example of two candidate hyperplanes +for our data sample. + +!split +===== Getting into the details ===== + +Let us define the function +!bt +\[ +f(x) = \bm{w}^T\bm{x}+b = 0, +\] +!et +as the function that determines the line $L$ that separates two classes (our two features), see the figure here. + + +Any point defined by $\bm{x}_i$ and $\bm{x}_2$ on the line $L$ will satisfy $\bm{w}^T(\bm{x}_1-\bm{x}_2)=0$. + +The signed distance $\delta$ from any point defined by a vector $\bm{x}$ and a point $\bm{x}_0$ on the line $L$ is then +!bt +\[ +\delta = \frac{1}{\vert\vert \bm{w}\vert\vert}(\bm{w}^T\bm{x}+b). +\] +!et + +!split +===== First attempt at a minimization approach ===== + +How do we find the parameter $b$ and the vector $\bm{w}$? What we could +do is to define a cost function which now contains the set of all +misclassified points $M$ and attempt to minimize this function + +!bt +\[ +C(\bm{w},b) = -\sum_{i\in M} y_i(\bm{w}^T\bm{x}_i+b). +\] +!et + +We could now for example define all values $y_i =1$ as misclassified in case we have $\bm{w}^T\bm{x}_i+b < 0$ and the opposite if we have $y_i=-1$. Taking the derivatives gives us +!bt +\[ +\frac{\partial C}{\partial b} = -\sum_{i\in M} y_i, +\] +!et +and +!bt +\[ +\frac{\partial C}{\partial \bm{w}} = -\sum_{i\in M} y_ix_i. +\] +!et + +!split +===== Solving the equations ===== + +We can now use the Newton-Raphson method or different variants of the gradient descent family (from plain gradient descent to various stochastic gradient descent approaches) to solve the equations +!bt +\[ +b \leftarrow b +\eta \frac{\partial C}{\partial b}, +\] +!et +and +!bt +\[ +\bm{w} \leftarrow \bm{w} +\eta \frac{\partial C}{\partial \bm{w}}, +\] +!et +where $\eta$ is our by now well-known learning rate. + + +!split +===== Code Example ===== + +The equations we discussed above can be coded rather easily (the +framework is similar to what we developed for logistic +regression). We are going to set up a simple case with two classes only and we want to find a line which separates them the best possible way. +!bc pycod + +!ec + +!split +===== Problems with the Simpler Approach ===== + + +There are however problems with this approach, although it looks +pretty straightforward to implement. When running the above code, we see that we can easily end up with many diffeent lines which separate the two classes. + + +For small +gaps between the entries, we may also end up needing many iterations +before the solutions converge and if the data cannot be separated +properly into two distinct classes, we may not experience a converge +at all. + +!split +===== A better approach ===== + +A better approach is rather to try to define a large margin between +the two classes (if they are well separated from the beginning). + +Thus, we wish to find a margin $M$ with $\bm{w}$ normalized to +$\vert\vert \bm{w}\vert\vert =1$ subject to the condition + +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, p. +\] +!et +All points are thus at a signed distance from the decision boundary defined by the line $L$. The parameters $b$ and $w_1$ and $w_2$ define this line. + +We seek thus the largest value $M$ defined by +!bt +\[ +\frac{1}{\vert \vert \bm{w}\vert\vert}y_i(\bm{w}^T\bm{x}_i+b) \geq M \hspace{0.1cm}\forall i=1,2,\dots, n, +\] +!et +or just +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b) \geq M\vert \vert \bm{w}\vert\vert \hspace{0.1cm}\forall i. +\] +!et +If we scale the equation so that $\vert \vert \bm{w}\vert\vert = 1/M$, we have to find the minimum of +$\bm{w}^T\bm{w}=\vert \vert \bm{w}\vert\vert$ (the norm) subject to the condition +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b) \geq 1 \hspace{0.1cm}\forall i. +\] +!et + +We have thus defined our margin as the invers of the norm of +$\bm{w}$. We want to minimize the norm in order to have a as large as +possible margin $M$. Before we proceed, we need to remind ourselves +about Lagrangian multipliers. + +!split +===== A quick Reminder on Lagrangian Multipliers ===== + +Consider a function of three independent variables $f(x,y,z)$ . For the function $f$ to be an +extreme we have +!bt +\[ +df=0. +\] +!et +A necessary and sufficient condition is +!bt +\[ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +\] +!et +due to +!bt +\[ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz. +\] +!et +In many problems the variables $x,y,z$ are often subject to constraints (such as those above for the margin) +so that they are no longer all independent. It is possible at least in principle to use each +constraint to eliminate one variable +and to proceed with a new and smaller set of independent varables. + +The use of so-called Lagrangian multipliers is an alternative technique when the elimination +of variables is incovenient or undesirable. Assume that we have an equation of constraint on +the variables $x,y,z$ +!bt +\[ +\phi(x,y,z) = 0, +\] +!et + resulting in +!bt +\[ +d\phi = \frac{\partial \phi}{\partial x}dx+\frac{\partial \phi}{\partial y}dy+\frac{\partial \phi}{\partial z}dz =0. +\] +!et +Now we cannot set anymore +!bt +\[ +\frac{\partial f}{\partial x} =\frac{\partial f}{\partial y}=\frac{\partial f}{\partial z}=0, +\] +!et +if $df=0$ is wanted +because there are now only two independent variables! Assume $x$ and $y$ are the independent +variables. +Then $dz$ is no longer arbitrary. + +!split +===== Adding the Multiplier ===== + +However, we can add to +!bt +\[ +df = \frac{\partial f}{\partial x}dx+\frac{\partial f}{\partial y}dy+\frac{\partial f}{\partial z}dz, +\] +!et +a multiplum of $d\phi$, viz. $\lambda d\phi$, resulting in +!bt +\[ +df+\lambda d\phi = (\frac{\partial f}{\partial z}+\lambda +\frac{\partial \phi}{\partial x})dx+(\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y})dy+ +(\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z})dz =0. +\] +!et +Our multiplier is chosen so that +!bt +\[ +\frac{\partial f}{\partial z}+\lambda\frac{\partial \phi}{\partial z} =0. +\] +!et + +We need to remember that we took $dx$ and $dy$ to be arbitrary and thus we must have +!bt +\[ +\frac{\partial f}{\partial x}+\lambda\frac{\partial \phi}{\partial x} =0, +\] +!et +and +!bt +\[ +\frac{\partial f}{\partial y}+\lambda\frac{\partial \phi}{\partial y} =0. +\] +!et +When all these equations are satisfied, $df=0$. We have four unknowns, $x,y,z$ and +$\lambda$. Actually we want only $x,y,z$, $\lambda$ needs not to be determined, +it is therefore often called +Lagrange's undetermined multiplier. +If we have a set of constraints $\phi_k$ we have the equations +!bt +\[ +\frac{\partial f}{\partial x_i}+\sum_k\lambda_k\frac{\partial \phi_k}{\partial x_i} =0. +\] +!et + +!split +===== Setting up the Problem ===== +In order to solve the above problem, we define the following Lagrangian function to be minimized +!bt +\[ +{\cal L}(\lambda,b,\bm{w})=\frac{1}{2}\bm{w}^T\bm{w}-\sum_{i=1}^n\lambda_i\left[y_i(\bm{w}^T\bm{x}_i+b)-1\right], +\] +!et +where $\lambda_i$ is a so-called Lagrange multiplier subject to the condition $\lambda_i \geq 0$. + +Taking the derivatives with respect to $b$ and $\bm{w}$ we obtain +!bt +\[ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +\] +!et +and +!bt +\[ +\frac{\partial {\cal L}}{\partial \bm{w}} = 0 = \bm{w}-\sum_{i} \lambda_iy_i\bm{x}_i. +\] +!et +Inserting these constraints into the equation for ${\cal L}$ we obtain +!bt +\[ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\bm{x}_i^T\bm{x}_j, +\] +!et +subject to the constraints $\lambda_i\geq 0$ and $\sum_i\lambda_iy_i=0$. +We must in addition satisfy the "Karush-Kuhn-Tucker":"https://en.wikipedia.org/wiki/Karush%E2%80%93Kuhn%E2%80%93Tucker_conditions" (KKT) condition +!bt +\[ +\lambda_i\left[y_i(\bm{w}^T\bm{x}_i+b) -1\right] \hspace{0.1cm}\forall i. +\] +!et +o If $\lambda_i > 0$, then $y_i(\bm{w}^T\bm{x}_i+b)=1$ and we say that $x_i$ is on the boundary. +o If $y_i(\bm{w}^T\bm{x}_i+b)> 1$, we say $x_i$ is not on the boundary and we set $\lambda_i=0$. +When $\lambda_i > 0$, the vectors $\bm{x}_i$ are called support vectors. They are the vectors closest to the line (or hyperplane) and define the margin $M$. + +!split +===== The problem to solve ===== + +We can rewrite +!bt +\[ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\bm{x}_i^T\bm{x}_j, +\] +!et +and its constraints in terms of a matrix-vector problem where we minimize w.r.t. $\lambda$ the following problem +!bt +\[ +\frac{1}{2} \bm{\lambda}^T\begin{bmatrix} y_1y_1\bm{x}_1^T\bm{x}_1 & y_1y_2\bm{x}_1^T\bm{x}_2 & \dots & \dots & y_1y_n\bm{x}_1^T\bm{x}_n \\ +y_2y_1\bm{x}_2^T\bm{x}_1 & y_2y_2\bm{x}_2^T\bm{x}_2 & \dots & \dots & y_1y_n\bm{x}_2^T\bm{x}_n \\ +\dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots \\ +y_ny_1\bm{x}_n^T\bm{x}_1 & y_ny_2\bm{x}_n^T\bm{x}_2 & \dots & \dots & y_ny_n\bm{x}_n^T\bm{x}_n \\ +\end{bmatrix}\bm{\lambda}-\mathbb{1}\bm{\lambda}, +\] +!et +subject to $\bm{y}^T\bm{\lambda}=0$. Here we defined the vectors $\bm{\lambda} =[\lambda_1,\lambda_2,\dots,\lambda_n]$ and +$\bm{y}=[y_1,y_2,\dots,y_n]$. + + +!split +===== The last steps ===== + +Solving the above problem, yields the values of $\lambda_i$. +To find the coefficients of your hyperplane we need simply to compute +!bt +\[ +\bm{w}=\sum_{i} \lambda_iy_i\bm{x}_i. +\] +!et +With our vector $\bm{w}$ we can in turn find the value of the intercept $b$ (here in two dimensions) via +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b)=1, +\] +!et +resulting in +!bt +\[ +b = \frac{1}{y_i}-\bm{w}^T\bm{x}_i, +\] +!et +or if we write it out in terms of the support vectors only, with $N_s$ being their number, we have +!bt +\[ +b = \frac{1}{N_s}\sum_{j\in N_s}\left(y_j-\sum_{i=1}^n\lambda_iy_i\bm{x}_i^T\bm{x}_j\right). +\] +!et +With our hyperplane coefficients we can use our classifier to assign any observation by simply using +!bt +\[ +y_i = \mathrm{sign}(\bm{w}^T\bm{x}_i+b). +\] +!et +Below we discuss how to find the optimal values of $\lambda_i$. Before we proceed however, we discuss now the so-called soft classifier. + +!split +===== A soft classifier ===== + +Till now, the margin is strictly defined by the support vectors. This defines what is called a hard classifier, that is the margins are well defined. + +Suppose now that classes overlap in feature space, as shown in the +figure here. One way to deal with this problem before we define the +so-called _kernel approach_, is to allow a kind of slack in the sense +that we allow some points to be on the wrong side of the margin. + +We introduce thus the so-called _slack_ variables $\bm{\xi} =[\xi_1,x_2,\dots,x_n]$ and +modify our previous equation +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b)=1, +\] +!et +to +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b)=1-\xi_i, +\] +!et +with the requirement $\xi_i\geq 0$. The total violation is now $\sum_i\xi$. +The value $\xi_i$ in the constraint the last constraint corresponds to the amount by which the prediction +$y_i(\bm{w}^T\bm{x}_i+b)=1$ is on the wrong side of its margin. Hence by bounding the sum $\sum_i \xi_i$, +we bound the total amount by which predictions fall on the wrong side of their margins. + +Misclassifications occur when $\xi_i > 1$. Thus bounding the total sum by some value $C$ bounds in turn the total number of +misclassifications. + +!split +===== Soft optmization problem ===== + + +This has in turn the consequences that we change our optmization problem to finding the minimum of +!bt +\[ +{\cal L}=\frac{1}{2}\bm{w}^T\bm{w}-\sum_{i=1}^n\lambda_i\left[y_i(\bm{w}^T\bm{x}_i+b)-(1-\xi_)\right]+C\sum_{i=1}^n\xi_i-\sum_{i=1}^n\gamma_i\xi_i, +\] +!et +subject to +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b)=1-\xi_i \hspace{0.1cm}\forall i, +\] +!et +with the requirement $\xi_i\geq 0$. + +Taking the derivatives with respect to $b$ and $\bm{w}$ we obtain +!bt +\[ +\frac{\partial {\cal L}}{\partial b} = -\sum_{i} \lambda_iy_i=0, +\] +!et +and +!bt +\[ +\frac{\partial {\cal L}}{\partial \bm{w}} = 0 = \bm{w}-\sum_{i} \lambda_iy_i\bm{x}_i, +\] +!et +and +!bt +\[ +\lambda_i = C-\gamma_i \hspace{0.1cm}\forall i. +\] +!et +Inserting these constraints into the equation for ${\cal L}$ we obtain the same equation as before +!bt +\[ +{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\bm{x}_i^T\bm{x}_j, +\] +!et +but now subject to the constraints $\lambda_i\geq 0$, $\sum_i\lambda_iy_i=0$ and $0\leq\lambda_i \leq C$. +We must in addition satisfy the Karush-Kuhn-Tucker condition which now reads +!bt +\[ +\lambda_i\left[y_i(\bm{w}^T\bm{x}_i+b) -(1-\xi_)\right]=0 \hspace{0.1cm}\forall i, +\] +!et +!bt +\[ +\gamma_i\xi_i = 0, +\] +!et +and +!bt +\[ +y_i(\bm{w}^T\bm{x}_i+b) -(1-\xi_) \geq 0 \hspace{0.1cm}\forall i. +\] +!et