941 lines
40 KiB
HTML
941 lines
40 KiB
HTML
|
||
<!DOCTYPE html>
|
||
|
||
|
||
<html lang="en" data-content_root="./" >
|
||
|
||
<head>
|
||
<meta charset="utf-8" />
|
||
<meta name="viewport" content="width=device-width, initial-scale=1.0" /><meta name="viewport" content="width=device-width, initial-scale=1" />
|
||
|
||
<title>12. Clustering and Unsupervised Learning — Applied Data Analysis and Machine Learning</title>
|
||
|
||
|
||
|
||
<script data-cfasync="false">
|
||
document.documentElement.dataset.mode = localStorage.getItem("mode") || "";
|
||
document.documentElement.dataset.theme = localStorage.getItem("theme") || "";
|
||
</script>
|
||
|
||
<!-- Loaded before other Sphinx assets -->
|
||
<link href="_static/styles/theme.css?digest=dfe6caa3a7d634c4db9b" rel="stylesheet" />
|
||
<link href="_static/styles/bootstrap.css?digest=dfe6caa3a7d634c4db9b" rel="stylesheet" />
|
||
<link href="_static/styles/pydata-sphinx-theme.css?digest=dfe6caa3a7d634c4db9b" rel="stylesheet" />
|
||
|
||
|
||
<link href="_static/vendor/fontawesome/6.5.2/css/all.min.css?digest=dfe6caa3a7d634c4db9b" rel="stylesheet" />
|
||
<link rel="preload" as="font" type="font/woff2" crossorigin href="_static/vendor/fontawesome/6.5.2/webfonts/fa-solid-900.woff2" />
|
||
<link rel="preload" as="font" type="font/woff2" crossorigin href="_static/vendor/fontawesome/6.5.2/webfonts/fa-brands-400.woff2" />
|
||
<link rel="preload" as="font" type="font/woff2" crossorigin href="_static/vendor/fontawesome/6.5.2/webfonts/fa-regular-400.woff2" />
|
||
|
||
<link rel="stylesheet" type="text/css" href="_static/pygments.css?v=fa44fd50" />
|
||
<link rel="stylesheet" type="text/css" href="_static/styles/sphinx-book-theme.css?v=eba8b062" />
|
||
<link rel="stylesheet" type="text/css" href="_static/togglebutton.css?v=13237357" />
|
||
<link rel="stylesheet" type="text/css" href="_static/copybutton.css?v=76b2166b" />
|
||
<link rel="stylesheet" type="text/css" href="_static/mystnb.8ecb98da25f57f5357bf6f572d296f466b2cfe2517ffebfabe82451661e28f02.css?v=6644e6bb" />
|
||
<link rel="stylesheet" type="text/css" href="_static/sphinx-thebe.css?v=4fa983c6" />
|
||
<link rel="stylesheet" type="text/css" href="_static/sphinx-design.min.css?v=95c83b7e" />
|
||
|
||
<!-- Pre-loaded scripts that we'll load fully later -->
|
||
<link rel="preload" as="script" href="_static/scripts/bootstrap.js?digest=dfe6caa3a7d634c4db9b" />
|
||
<link rel="preload" as="script" href="_static/scripts/pydata-sphinx-theme.js?digest=dfe6caa3a7d634c4db9b" />
|
||
<script src="_static/vendor/fontawesome/6.5.2/js/all.min.js?digest=dfe6caa3a7d634c4db9b"></script>
|
||
|
||
<script src="_static/documentation_options.js?v=9eb32ce0"></script>
|
||
<script src="_static/doctools.js?v=9a2dae69"></script>
|
||
<script src="_static/sphinx_highlight.js?v=dc90522c"></script>
|
||
<script src="_static/clipboard.min.js?v=a7894cd8"></script>
|
||
<script src="_static/copybutton.js?v=f281be69"></script>
|
||
<script src="_static/scripts/sphinx-book-theme.js?v=887ef09a"></script>
|
||
<script>let toggleHintShow = 'Click to show';</script>
|
||
<script>let toggleHintHide = 'Click to hide';</script>
|
||
<script>let toggleOpenOnPrint = 'true';</script>
|
||
<script src="_static/togglebutton.js?v=4a39c7ea"></script>
|
||
<script>var togglebuttonSelector = '.toggle, .admonition.dropdown';</script>
|
||
<script src="_static/design-tabs.js?v=f930bc37"></script>
|
||
<script>const THEBE_JS_URL = "https://unpkg.com/thebe@0.8.2/lib/index.js"; const thebe_selector = ".thebe,.cell"; const thebe_selector_input = "pre"; const thebe_selector_output = ".output, .cell_output"</script>
|
||
<script async="async" src="_static/sphinx-thebe.js?v=c100c467"></script>
|
||
<script>var togglebuttonSelector = '.toggle, .admonition.dropdown';</script>
|
||
<script>const THEBE_JS_URL = "https://unpkg.com/thebe@0.8.2/lib/index.js"; const thebe_selector = ".thebe,.cell"; const thebe_selector_input = "pre"; const thebe_selector_output = ".output, .cell_output"</script>
|
||
<script>window.MathJax = {"options": {"processHtmlClass": "tex2jax_process|mathjax_process|math|output_area"}}</script>
|
||
<script defer="defer" src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js"></script>
|
||
<script>DOCUMENTATION_OPTIONS.pagename = 'clustering';</script>
|
||
<link rel="index" title="Index" href="genindex.html" />
|
||
<link rel="search" title="Search" href="search.html" />
|
||
<link rel="next" title="13. Neural networks" href="chapter9.html" />
|
||
<link rel="prev" title="11. Basic ideas of the Principal Component Analysis (PCA)" href="chapter8.html" />
|
||
<meta name="viewport" content="width=device-width, initial-scale=1"/>
|
||
<meta name="docsearch:language" content="en"/>
|
||
</head>
|
||
|
||
|
||
<body data-bs-spy="scroll" data-bs-target=".bd-toc-nav" data-offset="180" data-bs-root-margin="0px 0px -60%" data-default-mode="">
|
||
|
||
|
||
|
||
<div id="pst-skip-link" class="skip-link d-print-none"><a href="#main-content">Skip to main content</a></div>
|
||
|
||
<div id="pst-scroll-pixel-helper"></div>
|
||
|
||
<button type="button" class="btn rounded-pill" id="pst-back-to-top">
|
||
<i class="fa-solid fa-arrow-up"></i>Back to top</button>
|
||
|
||
|
||
<input type="checkbox"
|
||
class="sidebar-toggle"
|
||
id="pst-primary-sidebar-checkbox"/>
|
||
<label class="overlay overlay-primary" for="pst-primary-sidebar-checkbox"></label>
|
||
|
||
<input type="checkbox"
|
||
class="sidebar-toggle"
|
||
id="pst-secondary-sidebar-checkbox"/>
|
||
<label class="overlay overlay-secondary" for="pst-secondary-sidebar-checkbox"></label>
|
||
|
||
<div class="search-button__wrapper">
|
||
<div class="search-button__overlay"></div>
|
||
<div class="search-button__search-container">
|
||
<form class="bd-search d-flex align-items-center"
|
||
action="search.html"
|
||
method="get">
|
||
<i class="fa-solid fa-magnifying-glass"></i>
|
||
<input type="search"
|
||
class="form-control"
|
||
name="q"
|
||
id="search-input"
|
||
placeholder="Search this book..."
|
||
aria-label="Search this book..."
|
||
autocomplete="off"
|
||
autocorrect="off"
|
||
autocapitalize="off"
|
||
spellcheck="false"/>
|
||
<span class="search-button__kbd-shortcut"><kbd class="kbd-shortcut__modifier">Ctrl</kbd>+<kbd>K</kbd></span>
|
||
</form></div>
|
||
</div>
|
||
|
||
<div class="pst-async-banner-revealer d-none">
|
||
<aside id="bd-header-version-warning" class="d-none d-print-none" aria-label="Version warning"></aside>
|
||
</div>
|
||
|
||
|
||
<header class="bd-header navbar navbar-expand-lg bd-navbar d-print-none">
|
||
</header>
|
||
|
||
|
||
<div class="bd-container">
|
||
<div class="bd-container__inner bd-page-width">
|
||
|
||
|
||
|
||
<div class="bd-sidebar-primary bd-sidebar">
|
||
|
||
|
||
|
||
<div class="sidebar-header-items sidebar-primary__section">
|
||
|
||
|
||
|
||
|
||
</div>
|
||
|
||
<div class="sidebar-primary-items__start sidebar-primary__section">
|
||
<div class="sidebar-primary-item">
|
||
|
||
|
||
|
||
|
||
|
||
<a class="navbar-brand logo" href="intro.html">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<img src="_static/logo.png" class="logo__image only-light" alt="Applied Data Analysis and Machine Learning - Home"/>
|
||
<script>document.write(`<img src="_static/logo.png" class="logo__image only-dark" alt="Applied Data Analysis and Machine Learning - Home"/>`);</script>
|
||
|
||
|
||
</a></div>
|
||
<div class="sidebar-primary-item">
|
||
|
||
<script>
|
||
document.write(`
|
||
<button class="btn search-button-field search-button__button" title="Search" aria-label="Search" data-bs-placement="bottom" data-bs-toggle="tooltip">
|
||
<i class="fa-solid fa-magnifying-glass"></i>
|
||
<span class="search-button__default-text">Search</span>
|
||
<span class="search-button__kbd-shortcut"><kbd class="kbd-shortcut__modifier">Ctrl</kbd>+<kbd class="kbd-shortcut__modifier">K</kbd></span>
|
||
</button>
|
||
`);
|
||
</script></div>
|
||
<div class="sidebar-primary-item"><nav class="bd-links bd-docs-nav" aria-label="Main">
|
||
<div class="bd-toc-item navbar-nav active">
|
||
|
||
<ul class="nav bd-sidenav bd-sidenav__home-link">
|
||
<li class="toctree-l1">
|
||
<a class="reference internal" href="intro.html">
|
||
Applied Data Analysis and Machine Learning
|
||
</a>
|
||
</li>
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">About the course</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="schedule.html">Course setting</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="teachers.html">Teachers and Grading</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="textbooks.html">Textbooks</a></li>
|
||
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">Review of Statistics with Resampling Techniques and Linear Algebra</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="statistics.html">1. Elements of Probability Theory and Statistical Data Analysis</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="linalg.html">2. Linear Algebra, Handling of Arrays and more Python Features</a></li>
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">From Regression to Support Vector Machines</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter1.html">3. Linear Regression</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter2.html">4. Ridge and Lasso Regression</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter3.html">5. Resampling Methods</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter4.html">6. Logistic Regression</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapteroptimization.html">7. Optimization, the central part of any Machine Learning algortithm</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter5.html">8. Support Vector Machines, overarching aims</a></li>
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">Decision Trees, Ensemble Methods and Boosting</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter6.html">9. Decision trees, overarching aims</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter7.html">10. Ensemble Methods: From a Single Tree to Many Trees and Extreme Boosting, Meet the Jungle of Methods</a></li>
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">Dimensionality Reduction</span></p>
|
||
<ul class="current nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter8.html">11. Basic ideas of the Principal Component Analysis (PCA)</a></li>
|
||
<li class="toctree-l1 current active"><a class="current reference internal" href="#">12. Clustering and Unsupervised Learning</a></li>
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">Deep Learning Methods</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter9.html">13. Neural networks</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter10.html">14. Building a Feed Forward Neural Network</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter11.html">15. Solving Differential Equations with Deep Learning</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter12.html">16. Convolutional Neural Networks</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="chapter13.html">17. Recurrent neural networks: Overarching view</a></li>
|
||
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">Weekly material, notes and exercises</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="exercisesweek34.html">Exercises week 34</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week34.html">Week 34: Introduction to the course, Logistics and Practicalities</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="exercisesweek35.html">Exercises week 35</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week35.html">Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="exercisesweek36.html">Exercises week 36</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week36.html">Week 36: Linear Regression and Gradient descent</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="exercisesweek37.html">Exercises week 37</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week37.html">Week 37: Gradient descent methods</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="exercisesweek38.html">Exercises week 38</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week38.html">Week 38: Statistical analysis, bias-variance tradeoff and resampling methods</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="exercisesweek39.html">Exercises week 39</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week39.html">Week 39: Resampling methods and logistic regression</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week40.html">Week 40: Gradient descent methods (continued) and start Neural networks</a></li>
|
||
<li class="toctree-l1"><a class="reference internal" href="week41.html">Week 41 Neural networks and constructing a neural network code</a></li>
|
||
</ul>
|
||
<p aria-level="2" class="caption" role="heading"><span class="caption-text">Projects</span></p>
|
||
<ul class="nav bd-sidenav">
|
||
<li class="toctree-l1"><a class="reference internal" href="project1.html">Project 1 on Machine Learning, deadline October 6 (midnight), 2025</a></li>
|
||
</ul>
|
||
|
||
</div>
|
||
</nav></div>
|
||
</div>
|
||
|
||
|
||
<div class="sidebar-primary-items__end sidebar-primary__section">
|
||
</div>
|
||
|
||
<div id="rtd-footer-container"></div>
|
||
|
||
|
||
</div>
|
||
|
||
<main id="main-content" class="bd-main" role="main">
|
||
|
||
|
||
|
||
<div class="sbt-scroll-pixel-helper"></div>
|
||
|
||
<div class="bd-content">
|
||
<div class="bd-article-container">
|
||
|
||
<div class="bd-header-article d-print-none">
|
||
<div class="header-article-items header-article__inner">
|
||
|
||
<div class="header-article-items__start">
|
||
|
||
<div class="header-article-item"><button class="sidebar-toggle primary-toggle btn btn-sm" title="Toggle primary sidebar" data-bs-placement="bottom" data-bs-toggle="tooltip">
|
||
<span class="fa-solid fa-bars"></span>
|
||
</button></div>
|
||
|
||
</div>
|
||
|
||
|
||
<div class="header-article-items__end">
|
||
|
||
<div class="header-article-item">
|
||
|
||
<div class="article-header-buttons">
|
||
|
||
|
||
|
||
|
||
|
||
<div class="dropdown dropdown-download-buttons">
|
||
<button class="btn dropdown-toggle" type="button" data-bs-toggle="dropdown" aria-expanded="false" aria-label="Download this page">
|
||
<i class="fas fa-download"></i>
|
||
</button>
|
||
<ul class="dropdown-menu">
|
||
|
||
|
||
|
||
<li><a href="_sources/clustering.ipynb" target="_blank"
|
||
class="btn btn-sm btn-download-source-button dropdown-item"
|
||
title="Download source file"
|
||
data-bs-placement="left" data-bs-toggle="tooltip"
|
||
>
|
||
|
||
|
||
<span class="btn__icon-container">
|
||
<i class="fas fa-file"></i>
|
||
</span>
|
||
<span class="btn__text-container">.ipynb</span>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<button onclick="window.print()"
|
||
class="btn btn-sm btn-download-pdf-button dropdown-item"
|
||
title="Print to PDF"
|
||
data-bs-placement="left" data-bs-toggle="tooltip"
|
||
>
|
||
|
||
|
||
<span class="btn__icon-container">
|
||
<i class="fas fa-file-pdf"></i>
|
||
</span>
|
||
<span class="btn__text-container">.pdf</span>
|
||
</button>
|
||
</li>
|
||
|
||
</ul>
|
||
</div>
|
||
|
||
|
||
|
||
|
||
<button onclick="toggleFullScreen()"
|
||
class="btn btn-sm btn-fullscreen-button"
|
||
title="Fullscreen mode"
|
||
data-bs-placement="bottom" data-bs-toggle="tooltip"
|
||
>
|
||
|
||
|
||
<span class="btn__icon-container">
|
||
<i class="fas fa-expand"></i>
|
||
</span>
|
||
|
||
</button>
|
||
|
||
|
||
|
||
<script>
|
||
document.write(`
|
||
<button class="btn btn-sm nav-link pst-navbar-icon theme-switch-button" title="light/dark" aria-label="light/dark" data-bs-placement="bottom" data-bs-toggle="tooltip">
|
||
<i class="theme-switch fa-solid fa-sun fa-lg" data-mode="light"></i>
|
||
<i class="theme-switch fa-solid fa-moon fa-lg" data-mode="dark"></i>
|
||
<i class="theme-switch fa-solid fa-circle-half-stroke fa-lg" data-mode="auto"></i>
|
||
</button>
|
||
`);
|
||
</script>
|
||
|
||
|
||
<script>
|
||
document.write(`
|
||
<button class="btn btn-sm pst-navbar-icon search-button search-button__button" title="Search" aria-label="Search" data-bs-placement="bottom" data-bs-toggle="tooltip">
|
||
<i class="fa-solid fa-magnifying-glass fa-lg"></i>
|
||
</button>
|
||
`);
|
||
</script>
|
||
<button class="sidebar-toggle secondary-toggle btn btn-sm" title="Toggle secondary sidebar" data-bs-placement="bottom" data-bs-toggle="tooltip">
|
||
<span class="fa-solid fa-list"></span>
|
||
</button>
|
||
</div></div>
|
||
|
||
</div>
|
||
|
||
</div>
|
||
</div>
|
||
|
||
|
||
|
||
<div id="jb-print-docs-body" class="onlyprint">
|
||
<h1>Clustering and Unsupervised Learning</h1>
|
||
<!-- Table of contents -->
|
||
<div id="print-main-content">
|
||
<div id="jb-print-toc">
|
||
|
||
<div>
|
||
<h2> Contents </h2>
|
||
</div>
|
||
<nav aria-label="Page">
|
||
<ul class="visible nav section-nav flex-column">
|
||
<li class="toc-h2 nav-item toc-entry"><a class="reference internal nav-link" href="#codes-and-approaches">12.1. Codes and Approaches</a></li>
|
||
</ul>
|
||
</nav>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
|
||
|
||
<div id="searchbox"></div>
|
||
<article class="bd-article">
|
||
|
||
<!-- HTML file automatically generated from DocOnce source (https://github.com/doconce/doconce/)
|
||
doconce format html clustering.do.txt --><section class="tex2jax_ignore mathjax_ignore" id="clustering-and-unsupervised-learning">
|
||
<h1><span class="section-number">12. </span>Clustering and Unsupervised Learning<a class="headerlink" href="#clustering-and-unsupervised-learning" title="Link to this heading">#</a></h1>
|
||
<p>In general terms cluster analysis, or clustering, is the task of grouping a
|
||
data-set into different distinct categories based on some measure of equality of
|
||
the data. This measure is often referred to as a <strong>metric</strong> or <strong>similarity
|
||
measure</strong> in the literature (note: sometimes we deal with a <strong>dissimilarity
|
||
measure</strong> instead). Usually, these metrics are formulated as some kind of
|
||
distance function between points in a high-dimensional space.</p>
|
||
<p>The simplest, and also the most
|
||
common is the <strong>Euclidean distance</strong>.</p>
|
||
<p>The simplest of all clustering algorithms is the <strong>k-means algorithm</strong>
|
||
, sometimes also referred to as <em>Lloyds algorithm</em>. It is the simplest and also
|
||
the most common. From its simplicity it obtains both strengths and weaknesses.
|
||
These will be discussed in more detail later. The <span class="math notranslate nohighlight">\(k\)</span>-means algorithm is a
|
||
<strong>centroid based</strong> clustering algorithm.</p>
|
||
<p>Assume, we are given <span class="math notranslate nohighlight">\(n\)</span> data points and we wish to split the data into <span class="math notranslate nohighlight">\(K < n\)</span>
|
||
different categories, or clusters. We label each cluster by an integer</p>
|
||
<div class="math notranslate nohighlight">
|
||
\[
|
||
k\in\{1, \cdots, K \}.
|
||
\]</div>
|
||
<p>In the basic k-means algorithm each point is assigned to only
|
||
one cluster <span class="math notranslate nohighlight">\(k\)</span>, and these assignments are <em>non-injective</em> i.e. many-to-one. We
|
||
can think of these mappings as an encoder <span class="math notranslate nohighlight">\(k = C(i)\)</span>, which assigns the <span class="math notranslate nohighlight">\(i\)</span>-th
|
||
data-point <span class="math notranslate nohighlight">\(\bf x_i\)</span> to the <span class="math notranslate nohighlight">\(k\)</span>-th cluster.</p>
|
||
<p><span class="math notranslate nohighlight">\(k\)</span>-means algorithm in words:</p>
|
||
<ol class="arabic simple">
|
||
<li><p>We start with guesses / random initializations of our <span class="math notranslate nohighlight">\(k\)</span> cluster centers/centroids</p></li>
|
||
<li><p>For each centroid the points that are most similar are identified</p></li>
|
||
<li><p>Then we move / replace each centroid with a coordinate average of all the points that were assigned to that centroid.</p></li>
|
||
<li><p>Iterate 2-3 until the centroids no longer move (to some tolerance)</p></li>
|
||
</ol>
|
||
<p>We assume we have <span class="math notranslate nohighlight">\(n\)</span> data-points</p>
|
||
<!-- Equation labels as ordinary links -->
|
||
<div id="eq:kmeanspoints"></div>
|
||
<div class="math notranslate nohighlight">
|
||
\[
|
||
\begin{equation}\label{eq:kmeanspoints} \tag{1}
|
||
\boldsymbol{x_i} = \{x_{i, 1}, \cdots, x_{i, p}\}\in\mathbb{R}^p.
|
||
\end{equation}
|
||
\]</div>
|
||
<p>which we wish to group into <span class="math notranslate nohighlight">\(K < n\)</span> clusters. For our dissimilarity measure we
|
||
use the <em>squared Euclidean distance</em></p>
|
||
<!-- Equation labels as ordinary links -->
|
||
<div id="eq:squaredeuclidean"></div>
|
||
<div class="math notranslate nohighlight">
|
||
\[
|
||
\begin{equation}\label{eq:squaredeuclidean} \tag{2}
|
||
d(\boldsymbol{x_i}, \boldsymbol{x_i'}) = \sum_{j=1}^p(x_{ij} - x_{i'j})^2
|
||
= ||\boldsymbol{x_i} - \boldsymbol{x_{i'}}||^2
|
||
\end{equation}
|
||
\]</div>
|
||
<p>We define the so called <em>within-cluster point scatter</em> which gives us a
|
||
measure of how close each data point assigned to the same cluster tends to be to
|
||
the all the others.</p>
|
||
<!-- Equation labels as ordinary links -->
|
||
<div id="eq:withincluster"></div>
|
||
<div class="math notranslate nohighlight">
|
||
\[
|
||
\begin{equation}\label{eq:withincluster} \tag{3}
|
||
W(C) = \frac{1}{2}\sum_{k=1}^K\sum_{C(i)=k}
|
||
\sum_{C(i')=k}d(\boldsymbol{x_i}, \boldsymbol{x_{i'}}) =
|
||
\sum_{k=1}^KN_k\sum_{C(i)=k}||\boldsymbol{x_i} - \boldsymbol{\overline{x_k}}||^2
|
||
\end{equation}
|
||
\]</div>
|
||
<p>where <span class="math notranslate nohighlight">\(\boldsymbol{\overline{x_k}}\)</span> is the mean vector associated with the <span class="math notranslate nohighlight">\(k\)</span>-th
|
||
cluster, and <span class="math notranslate nohighlight">\(N_k = \sum_{i=1}^nI(C(i) = k)\)</span>, where the <span class="math notranslate nohighlight">\(I()\)</span> notation is
|
||
similar to the Kronecker delta (<em>Commonly used in statistics, it just means that
|
||
when <span class="math notranslate nohighlight">\(i = k\)</span> we have the encoder <span class="math notranslate nohighlight">\(C(i)\)</span></em>). In other words, the within-cluster
|
||
scatter measures the compactness of each cluster with respect to the data points
|
||
assigned to each cluster. This is the quantity that the <span class="math notranslate nohighlight">\(k\)</span>-means algorithm aims
|
||
to minimize. We refer to this quantity <span class="math notranslate nohighlight">\(W(C)\)</span> as the within cluster scatter
|
||
because of its relation to the <em>total scatter</em>.</p>
|
||
<p>We have</p>
|
||
<!-- Equation labels as ordinary links -->
|
||
<div id="eq:totalscatter"></div>
|
||
<div class="math notranslate nohighlight">
|
||
\[
|
||
\begin{equation}\label{eq:totalscatter} \tag{4}
|
||
T = W(C) + B(C) = \frac{1}{2}\sum_{i=1}^n
|
||
\sum_{i'=1}^nd(\boldsymbol{x_i}, \boldsymbol{x_{i'}})
|
||
= \frac{1}{2}\sum_{k=1}^K\sum_{C(i)=k}
|
||
\Big(\sum_{C(i') = k}d(\boldsymbol{x_i}, \boldsymbol{x_{i'}})
|
||
+ \sum_{C(i')\neq k}d(\boldsymbol{x_i}, \boldsymbol{x_{i'}})\Big).
|
||
\end{equation}
|
||
\]</div>
|
||
<p>This is a quantity that is conserved throughout the <span class="math notranslate nohighlight">\(k\)</span>-means algorithm. It can
|
||
be thought of as the total amount of information in the data, and it is composed
|
||
of the aforementioned within-cluster scatter and the <em>between-cluster scatter</em>
|
||
<span class="math notranslate nohighlight">\(B(C)\)</span>. In methods such as principle component analysis the total scatter is not
|
||
conserved.</p>
|
||
<p>Given a cluster mean <span class="math notranslate nohighlight">\(\boldsymbol{m_k}\)</span> we define the <strong>total cluster variance</strong></p>
|
||
<!-- Equation labels as ordinary links -->
|
||
<div id="eq:totalclustervariance"></div>
|
||
<div class="math notranslate nohighlight">
|
||
\[
|
||
\begin{equation}\label{eq:totalclustervariance} \tag{5}
|
||
\min_{C, \{\boldsymbol{m_k}\}_1^K}\sum_{k=1}^KN_k\sum||\boldsymbol{x_i} - \boldsymbol{m_k}||^2
|
||
\end{equation}
|
||
\]</div>
|
||
<p>Now we have all the pieces necessary to formally revisit the <span class="math notranslate nohighlight">\(k\)</span>-means algorithm.</p>
|
||
<p>The <span class="math notranslate nohighlight">\(k\)</span>-means clustering algorithm goes as follows</p>
|
||
<ol class="arabic simple">
|
||
<li><p>For a given cluster assignment <span class="math notranslate nohighlight">\(C\)</span>, and <span class="math notranslate nohighlight">\(k\)</span> cluster means <span class="math notranslate nohighlight">\(\left\{m_1, \cdots, m_k\right\}\)</span>. We minimize the total cluster variance with respect to the cluster means <span class="math notranslate nohighlight">\(\{m_k\}\)</span> yielding the means of the currently assigned clusters.</p></li>
|
||
<li><p>Given a current set of <span class="math notranslate nohighlight">\(k\)</span> means <span class="math notranslate nohighlight">\(\{m_k\}\)</span> the total cluster variance is minimized by assigning each observation to the closest (current) cluster mean. That is $<span class="math notranslate nohighlight">\(C(i) = \underset{1\leq k\leq K}{\mathrm{argmin}} ||\boldsymbol{x_i} - \boldsymbol{m_k}||^2\)</span>$</p></li>
|
||
<li><p>Steps 1 and 2 are repeated until the assignments do not change.</p></li>
|
||
</ol>
|
||
<section id="codes-and-approaches">
|
||
<h2><span class="section-number">12.1. </span>Codes and Approaches<a class="headerlink" href="#codes-and-approaches" title="Link to this heading">#</a></h2>
|
||
<ol class="arabic simple">
|
||
<li><p>Before we start we specify a number <span class="math notranslate nohighlight">\(k\)</span> which is the number of clusters we want to try to separate our data into.</p></li>
|
||
<li><p>We initially choose <span class="math notranslate nohighlight">\(k\)</span> random data points in our data as our initial centroids, <em>or means</em> (this is where the name comes from).</p></li>
|
||
<li><p>Assign each data point to their closest centroid, based on the squared Euclidean distance.</p></li>
|
||
<li><p>For each of the <span class="math notranslate nohighlight">\(k\)</span> cluster we update the centroid by calculating new mean values for all the data points in the cluster.</p></li>
|
||
<li><p>Iteratively minimize the within cluster scatter by performing steps (3, 4) until the new assignments stop changing (can be to some tolerance) or until a maximum number of iterations have passed.</p></li>
|
||
</ol>
|
||
<p>Let us now program the most basic version of the algorithm using nothing but
|
||
Python with numpy arrays. This code is kept intentionally simple to gradually
|
||
progress our understanding. There is no vectorization of any kind, and even most
|
||
helper functions are not utilized.</p>
|
||
<p>We need first a dataset to do our cluster analysis on. In our case
|
||
this is a plain <em>vanilla</em> data set using random numbers using a
|
||
Gaussian distribution.</p>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>%matplotlib inline
|
||
|
||
import time
|
||
import numpy as np
|
||
import tensorflow as tf
|
||
from matplotlib import image
|
||
import matplotlib.pyplot as plt
|
||
from sklearn.cluster import KMeans
|
||
from IPython.display import display
|
||
|
||
np.random.seed(2021)
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<p>Next we define functions, for ease of use later, to generate Gaussians and to
|
||
set up our toy data set.</p>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>def gaussian_points(dim=2, n_points=1000, mean_vector=np.array([0, 0]),
|
||
sample_variance=1):
|
||
"""
|
||
Very simple custom function to generate gaussian distributed point clusters
|
||
with variable dimension, number of points, means in each direction
|
||
(must match dim) and sample variance.
|
||
|
||
Inputs:
|
||
dim (int)
|
||
n_points (int)
|
||
mean_vector (np.array) (where index 0 is x, index 1 is y etc.)
|
||
sample_variance (float)
|
||
|
||
Returns:
|
||
data (np.array): with dimensions (dim x n_points)
|
||
"""
|
||
|
||
mean_matrix = np.zeros(dim) + mean_vector
|
||
covariance_matrix = np.eye(dim) * sample_variance
|
||
data = np.random.multivariate_normal(mean_matrix, covariance_matrix,
|
||
n_points)
|
||
return data
|
||
|
||
|
||
|
||
def generate_simple_clustering_dataset(dim=2, n_points=1000, plotting=True,
|
||
return_data=True):
|
||
"""
|
||
Toy model to illustrate k-means clustering
|
||
"""
|
||
|
||
data1 = gaussian_points(mean_vector=np.array([5, 5]))
|
||
data2 = gaussian_points()
|
||
data3 = gaussian_points(mean_vector=np.array([1, 4.5]))
|
||
data4 = gaussian_points(mean_vector=np.array([5, 1]))
|
||
data = np.concatenate((data1, data2, data3, data4), axis=0)
|
||
|
||
if plotting:
|
||
fig, ax = plt.subplots()
|
||
ax.scatter(data[:, 0], data[:, 1], alpha=0.2)
|
||
ax.set_title('Toy Model Dataset')
|
||
plt.show()
|
||
|
||
|
||
if return_data:
|
||
return data
|
||
|
||
|
||
data = generate_simple_clustering_dataset()
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<p>With the above dataset we start
|
||
implementing the <span class="math notranslate nohighlight">\(k\)</span>-means algorithm.</p>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>
|
||
n_samples, dimensions = data.shape
|
||
n_clusters = 4
|
||
|
||
# we randomly initialize our centroids
|
||
np.random.seed(2021)
|
||
centroids = data[np.random.choice(n_samples, n_clusters, replace=False), :]
|
||
distances = np.zeros((n_samples, n_clusters))
|
||
|
||
# first we need to calculate the distance to each centroid from our data
|
||
for k in range(n_clusters):
|
||
for n in range(n_samples):
|
||
dist = 0
|
||
for d in range(dimensions):
|
||
dist += np.abs(data[n, d] - centroids[k, d])**2
|
||
distances[n, k] = dist
|
||
|
||
# we initialize an array to keep track of to which cluster each point belongs
|
||
# the way we set it up here the index tracks which point and the value which
|
||
# cluster the point belongs to
|
||
cluster_labels = np.zeros(n_samples, dtype='int')
|
||
|
||
# next we loop through our samples and for every point assign it to the cluster
|
||
# to which it has the smallest distance to
|
||
for n in range(n_samples):
|
||
# tracking variables (all of this is basically just an argmin)
|
||
smallest = 1e10
|
||
smallest_row_index = 1e10
|
||
for k in range(n_clusters):
|
||
if distances[n, k] < smallest:
|
||
smallest = distances[n, k]
|
||
smallest_row_index = k
|
||
|
||
cluster_labels[n] = smallest_row_index
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>fig = plt.figure()
|
||
ax = fig.add_subplot()
|
||
unique_cluster_labels = np.unique(cluster_labels)
|
||
for i in unique_cluster_labels:
|
||
ax.scatter(data[cluster_labels == i, 0],
|
||
data[cluster_labels == i, 1],
|
||
label = i,
|
||
alpha = 0.2)
|
||
ax.scatter(centroids[:, 0], centroids[:, 1], c='black')
|
||
|
||
ax.set_title("First Grouping of Points to Centroids")
|
||
|
||
plt.show()
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<p>So what do we have so far? We have ‘picked’ <span class="math notranslate nohighlight">\(k\)</span> centroids at random from our
|
||
data points. There are other ways of more intelligently choosing their
|
||
initializations, however for our purposes randomly is fine. Then we have
|
||
initialized an array ‘distances’ which holds the information of the distance,
|
||
<em>or dissimilarity</em>, of every point to of our centroids. Finally, we have
|
||
initialized an array ‘cluster_labels’ which according to our distances array
|
||
holds the information of to which centroid every point is assigned. This was the
|
||
first pass of our algorithm. Essentially, all we need to do now is repeat the
|
||
distance and assignment steps above until we have reached a desired convergence
|
||
or a maximum amount of iterations.</p>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>
|
||
max_iterations = 100
|
||
tolerance = 1e-8
|
||
|
||
for iteration in range(max_iterations):
|
||
prev_centroids = centroids.copy()
|
||
for k in range(n_clusters):
|
||
# this array will be used to update our centroid positions
|
||
vector_mean = np.zeros(dimensions)
|
||
mean_divisor = 0
|
||
for n in range(n_samples):
|
||
if cluster_labels[n] == k:
|
||
vector_mean += data[n, :]
|
||
mean_divisor += 1
|
||
|
||
# update according to the k means
|
||
centroids[k, :] = vector_mean / mean_divisor
|
||
|
||
# we find the dissimilarity
|
||
for k in range(n_clusters):
|
||
for n in range(n_samples):
|
||
dist = 0
|
||
for d in range(dimensions):
|
||
dist += np.abs(data[n, d] - centroids[k, d])**2
|
||
distances[n, k] = dist
|
||
|
||
# assign each point
|
||
for n in range(n_samples):
|
||
smallest = 1e10
|
||
smallest_row_index = 1e10
|
||
for k in range(n_clusters):
|
||
if distances[n, k] < smallest:
|
||
smallest = distances[n, k]
|
||
smallest_row_index = k
|
||
|
||
cluster_labels[n] = smallest_row_index
|
||
|
||
# convergence criteria
|
||
centroid_difference = np.sum(np.abs(centroids - prev_centroids))
|
||
if centroid_difference < tolerance:
|
||
print(f'Converged at iteration {iteration}')
|
||
break
|
||
|
||
elif iteration == max_iterations:
|
||
print(f'Did not converge in {max_iterations} iterations')
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<p>We now have a simple , un-optimized <span class="math notranslate nohighlight">\(k\)</span>-means
|
||
clustering implementation. Lets plot the final result</p>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>fig = plt.figure()
|
||
ax = fig.add_subplot()
|
||
unique_cluster_labels = np.unique(cluster_labels)
|
||
for i in unique_cluster_labels:
|
||
ax.scatter(data[cluster_labels == i, 0],
|
||
data[cluster_labels == i, 1],
|
||
label = i,
|
||
alpha = 0.2)
|
||
ax.scatter(centroids[:, 0], centroids[:, 1], c='black')
|
||
|
||
ax.set_title("Final Result of K-means Clustering")
|
||
|
||
plt.show()
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<div class="cell docutils container">
|
||
<div class="cell_input docutils container">
|
||
<div class="highlight-none notranslate"><div class="highlight"><pre><span></span>def naive_kmeans(data, n_clusters=4, max_iterations=100, tolerance=1e-8):
|
||
start_time = time.time()
|
||
|
||
n_samples, dimensions = data.shape
|
||
n_clusters = 4
|
||
#np.random.seed(2021)
|
||
centroids = data[np.random.choice(n_samples, n_clusters, replace=False), :]
|
||
distances = np.zeros((n_samples, n_clusters))
|
||
|
||
for k in range(n_clusters):
|
||
for n in range(n_samples):
|
||
dist = 0
|
||
for d in range(dimensions):
|
||
dist += np.abs(data[n, d] - centroids[k, d])**2
|
||
distances[n, k] = dist
|
||
|
||
cluster_labels = np.zeros(n_samples, dtype='int')
|
||
|
||
for n in range(n_samples):
|
||
smallest = 1e10
|
||
smallest_row_index = 1e10
|
||
for k in range(n_clusters):
|
||
if distances[n, k] < smallest:
|
||
smallest = distances[n, k]
|
||
smallest_row_index = k
|
||
|
||
cluster_labels[n] = smallest_row_index
|
||
|
||
for iteration in range(max_iterations):
|
||
prev_centroids = centroids.copy()
|
||
for k in range(n_clusters):
|
||
vector_mean = np.zeros(dimensions)
|
||
mean_divisor = 0
|
||
for n in range(n_samples):
|
||
if cluster_labels[n] == k:
|
||
vector_mean += data[n, :]
|
||
mean_divisor += 1
|
||
|
||
centroids[k, :] = vector_mean / mean_divisor
|
||
|
||
for k in range(n_clusters):
|
||
for n in range(n_samples):
|
||
dist = 0
|
||
for d in range(dimensions):
|
||
dist += np.abs(data[n, d] - centroids[k, d])**2
|
||
distances[n, k] = dist
|
||
|
||
for n in range(n_samples):
|
||
smallest = 1e10
|
||
smallest_row_index = 1e10
|
||
for k in range(n_clusters):
|
||
if distances[n, k] < smallest:
|
||
smallest = distances[n, k]
|
||
smallest_row_index = k
|
||
|
||
cluster_labels[n] = smallest_row_index
|
||
|
||
centroid_difference = np.sum(np.abs(centroids - prev_centroids))
|
||
if centroid_difference < tolerance:
|
||
print(f'Converged at iteration {iteration}')
|
||
print(f'Runtime: {time.time() - start_time} seconds')
|
||
|
||
return cluster_labels, centroids
|
||
|
||
print(f'Did not converge in {max_iterations} iterations')
|
||
print(f'Runtime: {time.time() - start_time} seconds')
|
||
|
||
return cluster_labels, centroids
|
||
</pre></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</section>
|
||
</section>
|
||
|
||
<script type="text/x-thebe-config">
|
||
{
|
||
requestKernel: true,
|
||
binderOptions: {
|
||
repo: "binder-examples/jupyter-stacks-datascience",
|
||
ref: "master",
|
||
},
|
||
codeMirrorConfig: {
|
||
theme: "abcdef",
|
||
mode: "python"
|
||
},
|
||
kernelOptions: {
|
||
name: "python3",
|
||
path: "./."
|
||
},
|
||
predefinedOutput: true
|
||
}
|
||
</script>
|
||
<script>kernelName = 'python3'</script>
|
||
|
||
</article>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<footer class="prev-next-footer d-print-none">
|
||
|
||
<div class="prev-next-area">
|
||
<a class="left-prev"
|
||
href="chapter8.html"
|
||
title="previous page">
|
||
<i class="fa-solid fa-angle-left"></i>
|
||
<div class="prev-next-info">
|
||
<p class="prev-next-subtitle">previous</p>
|
||
<p class="prev-next-title"><span class="section-number">11. </span>Basic ideas of the Principal Component Analysis (PCA)</p>
|
||
</div>
|
||
</a>
|
||
<a class="right-next"
|
||
href="chapter9.html"
|
||
title="next page">
|
||
<div class="prev-next-info">
|
||
<p class="prev-next-subtitle">next</p>
|
||
<p class="prev-next-title"><span class="section-number">13. </span>Neural networks</p>
|
||
</div>
|
||
<i class="fa-solid fa-angle-right"></i>
|
||
</a>
|
||
</div>
|
||
</footer>
|
||
|
||
</div>
|
||
|
||
|
||
|
||
<div class="bd-sidebar-secondary bd-toc"><div class="sidebar-secondary-items sidebar-secondary__inner">
|
||
|
||
|
||
<div class="sidebar-secondary-item">
|
||
<div class="page-toc tocsection onthispage">
|
||
<i class="fa-solid fa-list"></i> Contents
|
||
</div>
|
||
<nav class="bd-toc-nav page-toc">
|
||
<ul class="visible nav section-nav flex-column">
|
||
<li class="toc-h2 nav-item toc-entry"><a class="reference internal nav-link" href="#codes-and-approaches">12.1. Codes and Approaches</a></li>
|
||
</ul>
|
||
</nav></div>
|
||
|
||
</div></div>
|
||
|
||
|
||
</div>
|
||
<footer class="bd-footer-content">
|
||
|
||
<div class="bd-footer-content__inner container">
|
||
|
||
<div class="footer-item">
|
||
|
||
<p class="component-author">
|
||
By Morten Hjorth-Jensen
|
||
</p>
|
||
|
||
</div>
|
||
|
||
<div class="footer-item">
|
||
|
||
|
||
<p class="copyright">
|
||
|
||
© Copyright 2023.
|
||
<br/>
|
||
|
||
</p>
|
||
|
||
</div>
|
||
|
||
<div class="footer-item">
|
||
|
||
</div>
|
||
|
||
<div class="footer-item">
|
||
|
||
</div>
|
||
|
||
</div>
|
||
</footer>
|
||
|
||
|
||
</main>
|
||
</div>
|
||
</div>
|
||
|
||
<!-- Scripts loaded after <body> so the DOM is not blocked -->
|
||
<script src="_static/scripts/bootstrap.js?digest=dfe6caa3a7d634c4db9b"></script>
|
||
<script src="_static/scripts/pydata-sphinx-theme.js?digest=dfe6caa3a7d634c4db9b"></script>
|
||
|
||
<footer class="bd-footer">
|
||
</footer>
|
||
</body>
|
||
</html> |