updating lectures

This commit is contained in:
mhjensen
2020-09-14 10:52:04 +02:00
parent 33fc031bd8
commit 490fdaaefb
41 changed files with 18995 additions and 5 deletions
+19 -1
View File
@@ -346,7 +346,25 @@
]
}
],
"metadata": {},
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.8.3"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
+90
View File
@@ -0,0 +1,90 @@
3.3773726001100143E-005, 3.1715032621225665E-002
2.7018980800880114E-004, 0.25379346864959029
9.1189060202970370E-004, 0.85691856357595775
2.1615184640704091E-003, 2.0322700794292126
4.2217157501375172E-003, 3.9716946611295496
7.2951248162376296E-003, 6.8678138410008760
1.1584388018377344E-002, 10.913973030767592
1.7292147712563273E-002, 16.304871533080519
2.4621046254802003E-002, 23.235226466525493
3.3773726001100138E-002, 31.902701432416372
4.4952829307464297E-002, 42.504284679798289
5.8360998529901037E-002, 55.243642496340065
7.4200876024417023E-002, 70.325036648868448
9.2675104147018753E-002, 87.967299710326188
0.11398632525371297, 108.39264632432710
0.13833718170050618 , 131.84282952216495
0.16593031584340495 , 158.57124952965228
0.19696837003841602 , 188.82820513039681
0.203199995458126059, 195.042764094891368
0.211199995279312130, 202.937172130123315
0.219199995100498229, 210.852580165355278
0.227199994921684245, 218.789988200587175
0.235199994742870316, 226.748396235819115
0.243199994564056388, 234.740804271051104
0.251199994385242487, 242.757212306283037
0.259199994206428530, 250.797620341515000
0.267199994027614574, 258.874028376746878
0.275199993848800673, 266.977436411978829
0.283199993669986716, 275.092844447210780
0.291199993491172815, 283.248252482442751
0.299199993312358858, 291.445660517674696
0.307199993133544902, 299.655068552906641
0.315199992954731001, 307.909476588138546
0.323199992775917044, 316.192884623370503
0.339199992418289187, 332.846700693834407
0.355199992060661329, 349.620516764298316
0.371199991703033416, 366.537332834762140
0.387199991345405559, 383.604148905225998
0.403199990987777701, 400.801964975689941
0.419199990630149844, 418.158781046153820
0.435199990272521986, 435.624597116617792
0.451199989914894073, 453.288413187081574
0.467199989557266215, 471.062229257545425
0.483199989199638358, 489.048045328009380
0.499199988842010500, 507.146861398473277
0.515199988484382643, 525.392677468937222
0.531199988126754730, 543.829493539401028
0.531999988108873390, 544.796634342924222
0.550079987704753859, 565.860416502548446
0.567999987304210641, 586.865970501467928
0.585599986910820047, 607.703068178978242
0.603199986517429343, 628.645165856488575
0.620639986127614951, 649.578035373294142
0.637919985741376872, 670.496676729395176
0.655199985355138792, 691.549318085496111
0.672479984968900713, 712.766959441597237
0.689599984586238834, 733.981372636993456
0.706879984200000755, 755.521013993094471
0.724159983813762675, 777.228655349195492
0.741439983427524596, 799.117296705296553
0.758719983041286516, 821.191938061397536
0.775999982655048326, 843.460579417498366
0.793439982265233934, 866.080448934304059
0.810879981875419542, 888.907318451109631
0.828479981482028949, 912.100416128620054
0.846239981085061932, 935.664741966834868
0.864159980684518825, 959.607295965754474
0.882079980283975607, 983.787849964674024
0.900159979879856187, 1008.36063212429838
0.918399979472160344, 1033.33564244462718
0.936639979064464612, 1058.56865276495591
0.955199978649616255, 1084.36811940669395
0.973919978231191585, 1110.59381420913678
0.992799977809190715, 1137.25073717228429
1.01183997738361353, 1164.35288829613614
1.06047997629642499, 1234.41724915034661
1.11039997518062594, 1307.72443529019392
1.16191997402906422, 1384.74890303708753
1.21535997283458719, 1466.00110871243692
1.27071997159719463, 1551.73305231624181
1.32847997030615828, 1642.68741833061654
1.38895996895432461, 1739.56266307697001
1.45263996753096580, 1843.32947103741640
1.52031996601820008, 1955.19398301547858
1.59295996439456933, 2077.14256797538474
1.67167996263504026, 2211.44382304206692
1.75855996069312082, 2361.74471430468566
1.85647995850443825, 2533.64534865592441
1.96959995597600934, 2734.93465827410455
2.10543995293974895, 2980.03336671234320
1 3.3773726001100143E-005 3.1715032621225665E-002
2 2.7018980800880114E-004 0.25379346864959029
3 9.1189060202970370E-004 0.85691856357595775
4 2.1615184640704091E-003 2.0322700794292126
5 4.2217157501375172E-003 3.9716946611295496
6 7.2951248162376296E-003 6.8678138410008760
7 1.1584388018377344E-002 10.913973030767592
8 1.7292147712563273E-002 16.304871533080519
9 2.4621046254802003E-002 23.235226466525493
10 3.3773726001100138E-002 31.902701432416372
11 4.4952829307464297E-002 42.504284679798289
12 5.8360998529901037E-002 55.243642496340065
13 7.4200876024417023E-002 70.325036648868448
14 9.2675104147018753E-002 87.967299710326188
15 0.11398632525371297 108.39264632432710
16 0.13833718170050618 131.84282952216495
17 0.16593031584340495 158.57124952965228
18 0.19696837003841602 188.82820513039681
19 0.203199995458126059 195.042764094891368
20 0.211199995279312130 202.937172130123315
21 0.219199995100498229 210.852580165355278
22 0.227199994921684245 218.789988200587175
23 0.235199994742870316 226.748396235819115
24 0.243199994564056388 234.740804271051104
25 0.251199994385242487 242.757212306283037
26 0.259199994206428530 250.797620341515000
27 0.267199994027614574 258.874028376746878
28 0.275199993848800673 266.977436411978829
29 0.283199993669986716 275.092844447210780
30 0.291199993491172815 283.248252482442751
31 0.299199993312358858 291.445660517674696
32 0.307199993133544902 299.655068552906641
33 0.315199992954731001 307.909476588138546
34 0.323199992775917044 316.192884623370503
35 0.339199992418289187 332.846700693834407
36 0.355199992060661329 349.620516764298316
37 0.371199991703033416 366.537332834762140
38 0.387199991345405559 383.604148905225998
39 0.403199990987777701 400.801964975689941
40 0.419199990630149844 418.158781046153820
41 0.435199990272521986 435.624597116617792
42 0.451199989914894073 453.288413187081574
43 0.467199989557266215 471.062229257545425
44 0.483199989199638358 489.048045328009380
45 0.499199988842010500 507.146861398473277
46 0.515199988484382643 525.392677468937222
47 0.531199988126754730 543.829493539401028
48 0.531999988108873390 544.796634342924222
49 0.550079987704753859 565.860416502548446
50 0.567999987304210641 586.865970501467928
51 0.585599986910820047 607.703068178978242
52 0.603199986517429343 628.645165856488575
53 0.620639986127614951 649.578035373294142
54 0.637919985741376872 670.496676729395176
55 0.655199985355138792 691.549318085496111
56 0.672479984968900713 712.766959441597237
57 0.689599984586238834 733.981372636993456
58 0.706879984200000755 755.521013993094471
59 0.724159983813762675 777.228655349195492
60 0.741439983427524596 799.117296705296553
61 0.758719983041286516 821.191938061397536
62 0.775999982655048326 843.460579417498366
63 0.793439982265233934 866.080448934304059
64 0.810879981875419542 888.907318451109631
65 0.828479981482028949 912.100416128620054
66 0.846239981085061932 935.664741966834868
67 0.864159980684518825 959.607295965754474
68 0.882079980283975607 983.787849964674024
69 0.900159979879856187 1008.36063212429838
70 0.918399979472160344 1033.33564244462718
71 0.936639979064464612 1058.56865276495591
72 0.955199978649616255 1084.36811940669395
73 0.973919978231191585 1110.59381420913678
74 0.992799977809190715 1137.25073717228429
75 1.01183997738361353 1164.35288829613614
76 1.06047997629642499 1234.41724915034661
77 1.11039997518062594 1307.72443529019392
78 1.16191997402906422 1384.74890303708753
79 1.21535997283458719 1466.00110871243692
80 1.27071997159719463 1551.73305231624181
81 1.32847997030615828 1642.68741833061654
82 1.38895996895432461 1739.56266307697001
83 1.45263996753096580 1843.32947103741640
84 1.52031996601820008 1955.19398301547858
85 1.59295996439456933 2077.14256797538474
86 1.67167996263504026 2211.44382304206692
87 1.75855996069312082 2361.74471430468566
88 1.85647995850443825 2533.64534865592441
89 1.96959995597600934 2734.93465827410455
90 2.10543995293974895 2980.03336671234320
File diff suppressed because it is too large Load Diff
Binary file not shown.

After

Width:  |  Height:  |  Size: 17 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 16 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 12 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 21 KiB

File diff suppressed because one or more lines are too long
File diff suppressed because it is too large Load Diff
Binary file not shown.

After

Width:  |  Height:  |  Size: 15 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 19 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 25 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 21 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 44 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 10 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 10 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 9.3 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 14 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 21 KiB

File diff suppressed because one or more lines are too long
Binary file not shown.

Before

Width:  |  Height:  |  Size: 111 KiB

After

Width:  |  Height:  |  Size: 111 KiB

File diff suppressed because one or more lines are too long
File diff suppressed because it is too large Load Diff
Binary file not shown.

After

Width:  |  Height:  |  Size: 15 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 24 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 95 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 34 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 10 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 9.1 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 25 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 11 KiB

@@ -0,0 +1,64 @@
Traceback (most recent call last):
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/jupyter_cache/executors/utils.py", line 51, in single_nb_execution
executenb(
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 1082, in execute
return NotebookClient(nb=nb, resources=resources, km=km, **kwargs).execute()
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 74, in wrapped
return just_run(coro(*args, **kwargs))
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 53, in just_run
return loop.run_until_complete(coro)
File "/Users/hjensen/opt/anaconda3/lib/python3.8/asyncio/base_events.py", line 616, in run_until_complete
return future.result()
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 535, in async_execute
await self.async_execute_cell(
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 827, in async_execute_cell
self._check_raise_for_error(cell, exec_reply)
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 735, in _check_raise_for_error
raise CellExecutionError.from_cell_and_msg(cell, exec_reply['content'])
nbclient.exceptions.CellExecutionError: An error occurred while executing the following cell:
------------------
# Common imports
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import sklearn.linear_model as skl
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
import os
# Where to save the figures and data files
PROJECT_ROOT_DIR = "Results"
FIGURE_ID = "Results/FigureFiles"
DATA_ID = "DataFiles/"
if not os.path.exists(PROJECT_ROOT_DIR):
os.mkdir(PROJECT_ROOT_DIR)
if not os.path.exists(FIGURE_ID):
os.makedirs(FIGURE_ID)
if not os.path.exists(DATA_ID):
os.makedirs(DATA_ID)
def image_path(fig_id):
return os.path.join(FIGURE_ID, fig_id)
def data_path(dat_id):
return os.path.join(DATA_ID, dat_id)
def save_fig(fig_id):
plt.savefig(image_path(fig_id) + ".png", format='png')
infile = open(data_path("MassEval2016.dat"),'r')
------------------
---------------------------------------------------------------------------
FileNotFoundError Traceback (most recent call last)
<ipython-input-28-3cd19a0768e1> in <module>
 31 plt.savefig(image_path(fig_id) + ".png", format='png')
 32 
---> 33 infile = open(data_path("MassEval2016.dat"),'r')

FileNotFoundError: [Errno 2] No such file or directory: 'DataFiles/MassEval2016.dat'
FileNotFoundError: [Errno 2] No such file or directory: 'DataFiles/MassEval2016.dat'
@@ -0,0 +1,116 @@
Traceback (most recent call last):
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/jupyter_cache/executors/utils.py", line 51, in single_nb_execution
executenb(
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 1082, in execute
return NotebookClient(nb=nb, resources=resources, km=km, **kwargs).execute()
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 74, in wrapped
return just_run(coro(*args, **kwargs))
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 53, in just_run
return loop.run_until_complete(coro)
File "/Users/hjensen/opt/anaconda3/lib/python3.8/asyncio/base_events.py", line 616, in run_until_complete
return future.result()
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 535, in async_execute
await self.async_execute_cell(
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 827, in async_execute_cell
self._check_raise_for_error(cell, exec_reply)
File "/Users/hjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 735, in _check_raise_for_error
raise CellExecutionError.from_cell_and_msg(cell, exec_reply['content'])
nbclient.exceptions.CellExecutionError: An error occurred while executing the following cell:
------------------
from numpy import *
from numpy.random import randint, randn
from time import time
import matplotlib.mlab as mlab
import matplotlib.pyplot as plt
# Returns mean of bootstrap samples
def stat(data):
return mean(data)
# Bootstrap algorithm
def bootstrap(data, statistic, R):
t = zeros(R); n = len(data); inds = arange(n); t0 = time()
# non-parametric bootstrap
for i in range(R):
t[i] = statistic(data[randint(0,n,n)])
# analysis
print("Runtime: %g sec" % (time()-t0)); print("Bootstrap Statistics :")
print("original bias std. error")
print("%8g %8g %14g %15g" % (statistic(data), std(data),mean(t),std(t)))
return t
mu, sigma = 100, 15
datapoints = 10000
x = mu + sigma*random.randn(datapoints)
# bootstrap returns the data sample
t = bootstrap(x, stat, datapoints)
# the histogram of the bootstrapped data
n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75)
# add a 'best fit' line
y = mlab.normpdf( binsboot, mean(t), std(t))
lt = plt.plot(binsboot, y, 'r--', linewidth=1)
plt.xlabel('Smarts')
plt.ylabel('Probability')
plt.axis([99.5, 100.6, 0, 3.0])
plt.grid(True)
plt.show()
------------------
---------------------------------------------------------------------------
AttributeError Traceback (most recent call last)
<ipython-input-29-53990135e988> in <module>
 29 t = bootstrap(x, stat, datapoints)
 30 # the histogram of the bootstrapped data
---> 31 n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75)
 32 
 33 # add a 'best fit' line
~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/pyplot.py in hist(x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, data, **kwargs)
 2603 orientation='vertical', rwidth=None, log=False, color=None,
 2604 label=None, stacked=False, *, data=None, **kwargs):
-> 2605 return gca().hist(
 2606 x, bins=bins, range=range, density=density, weights=weights,
 2607 cumulative=cumulative, bottom=bottom, histtype=histtype,
~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/__init__.py in inner(ax, data, *args, **kwargs)
 1563 def inner(ax, *args, data=None, **kwargs):
 1564 if data is None:
-> 1565 return func(ax, *map(sanitize_sequence, args), **kwargs)
 1566 
 1567 bound = new_sig.bind(ax, *args, **kwargs)
~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/axes/_axes.py in hist(self, x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, **kwargs)
 6817 if patch:
 6818 p = patch[0]
-> 6819 p.update(kwargs)
 6820 if lbl is not None:
 6821 p.set_label(lbl)
~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/artist.py in update(self, props)
 1004 
 1005 with cbook._setattr_cm(self, eventson=False):
-> 1006 ret = [_update_property(self, k, v) for k, v in props.items()]
 1007 
 1008 if len(ret):
~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/artist.py in <listcomp>(.0)
 1004 
 1005 with cbook._setattr_cm(self, eventson=False):
-> 1006 ret = [_update_property(self, k, v) for k, v in props.items()]
 1007 
 1008 if len(ret):
~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/artist.py in _update_property(self, k, v)
 999 func = getattr(self, 'set_' + k, None)
 1000 if not callable(func):
-> 1001 raise AttributeError('{!r} object has no property {!r}'
 1002 .format(type(self).__name__, k))
 1003 return func(v)
AttributeError: 'Rectangle' object has no property 'normed'
AttributeError: 'Rectangle' object has no property 'normed'
+1 -1
View File
@@ -1,4 +1,4 @@
- file: introduction.ipynb
- file: intro.md
- part: Getting started #includes also linear algebra
chapters:
- file: gettingstarted.ipynb
+339
View File
@@ -0,0 +1,339 @@
<!-- dom:TITLE: Introduction to Applied Data Analysis and Machine Learning -->
# Introduction to Applied Data Analysis and Machine Learning
<!-- dom:AUTHOR: Morten Hjorth-Jensen at Department of Physics, University of Oslo & Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University -->
<!-- Author: -->
**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University
Date: **Nov 19, 2019**
Copyright 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
## Introduction
During the last two decades there has been a swift and amazing
development of Machine Learning techniques and algorithms that impact
many areas in not only Science and Technology but also the Humanities,
Social Sciences, Medicine, Law, indeed, almost all possible
disciplines. The applications are incredibly many, from self-driving
cars to solving high-dimensional differential equations or complicated
quantum mechanical many-body problems. Machine Learning is perceived
by many as one of the main disruptive techniques nowadays.
Statistics, Data science and Machine Learning form important
fields of research in modern science. They describe how to learn and
make predictions from data, as well as allowing us to extract
important correlations about physical process and the underlying laws
of motion in large data sets. The latter, big data sets, appear
frequently in essentially all disciplines, from the traditional
Science, Technology, Mathematics and Engineering fields to Life
Science, Law, education research, the Humanities and the Social
Sciences.
It has become more
and more common to see research projects on big data in for example
the Social Sciences where extracting patterns from complicated survey
data is one of many research directions. Having a solid grasp of data
analysis and machine learning is thus becoming central to scientific
computing in many fields, and competences and skills within the fields
of machine learning and scientific computing are nowadays strongly
requested by many potential employers. The latter cannot be
overstated, familiarity with machine learning has almost become a
prerequisite for many of the most exciting employment opportunities,
whether they are in bioinformatics, life science, physics or finance,
in the private or the public sector. This author has had several
students or met students who have been hired recently based on their
skills and competences in scientific computing and data science, often
with marginal knowledge of machine learning.
Machine learning is a subfield of computer science, and is closely
related to computational statistics. It evolved from the study of
pattern recognition in artificial intelligence (AI) research, and has
made contributions to AI tasks like computer vision, natural language
processing and speech recognition. Many of the methods we will study are also
strongly rooted in basic mathematics and physics research.
Ideally, machine learning represents the science of giving computers
the ability to learn without being explicitly programmed. The idea is
that there exist generic algorithms which can be used to find patterns
in a broad class of data sets without having to write code
specifically for each problem. The algorithm will build its own logic
based on the data. You should however always keep in mind that
machines and algorithms are to a large extent developed by humans. The
insights and knowledge we have about a specific system, play a central
role when we develop a specific machine learning algorithm.
Machine learning is an extremely rich field, in spite of its young
age. The increases we have seen during the last three decades in
computational capabilities have been followed by developments of
methods and techniques for analyzing and handling large date sets,
relying heavily on statistics, computer science and mathematics. The
field is rather new and developing rapidly. Popular software packages
written in Python for machine learning like
[Scikit-learn](http://scikit-learn.org/stable/),
[Tensorflow](https://www.tensorflow.org/),
[PyTorch](http://pytorch.org/) and [Keras](https://keras.io/), all
freely available at their respective GitHub sites, encompass
communities of developers in the thousands or more. And the number of
code developers and contributors keeps increasing. Not all the
algorithms and methods can be given a rigorous mathematical
justification, opening up thereby large rooms for experimenting and
trial and error and thereby exciting new developments. However, a
solid command of linear algebra, multivariate theory, probability
theory, statistical data analysis, understanding errors and Monte
Carlo methods are central elements in a proper understanding of many
of algorithms and methods we will discuss.
<!-- !split -->
## Learning outcomes
These sets of lectures aim at giving you an overview of central aspects of
statistical data analysis as well as some of the central algorithms
used in machine learning. We will introduce a variety of central
algorithms and methods essential for studies of data analysis and
machine learning.
Hands-on projects and experimenting with data and algorithms plays a central role in
these lectures, and our hope is, through the various
projects and exercises, to expose you to fundamental
research problems in these fields, with the aim to reproduce state of
the art scientific results. You will learn to develop and
structure codes for studying these systems, get acquainted with
computing facilities and learn to handle large scientific projects. A
good scientific and ethical conduct is emphasized throughout the
course. More specifically, you will
1. Learn about basic data analysis, Bayesian statistics, Monte Carlo methods, data optimization and machine learning;
2. Be capable of extending the acquired knowledge to other systems and cases;
3. Have an understanding of central algorithms used in data analysis and machine learning;
4. Gain knowledge of central aspects of Monte Carlo methods, Markov chains, Gibbs samplers and their possible applications, from numerical integration to simulation of stock markets;
5. Understand methods for regression and classification;
6. Learn about neural network, genetic algorithms and Boltzmann machines;
7. Work on numerical projects to illustrate the theory. The projects play a central role and you are expected to know modern programming languages like Python or C++, in addition to a basic knowledge of linear algebra (typically taught during the first one or two years of undergraduate studies).
There are several topics we will cover here, spanning from
statistical data analysis and its basic concepts such as expectation
values, variance, covariance, correlation functions and errors, via
well-known probability distribution functions like the uniform
distribution, the binomial distribution, the Poisson distribution and
simple and multivariate normal distributions to central elements of
Bayesian statistics and modeling. We will also remind the reader about
central elements from linear algebra and standard methods based on
linear algebra used to optimize (minimize) functions (the family of gradient descent methods)
and the Singular-value decomposition and
least square methods for parameterizing data.
We will also cover Monte Carlo methods, Markov chains, well-known
algorithms for sampling stochastic events like the Metropolis-Hastings
and Gibbs sampling methods. An important aspect of all our
calculations is a proper estimation of errors. Here we will also
discuss famous resampling techniques like the blocking, the bootstrapping
and the jackknife methods and the infamous bias-variance tradeoff.
The second part of the material covers several algorithms used in
machine learning.
## Types of Machine Learning
The approaches to machine learning are many, but are often split into
two main categories. In *supervised learning* we know the answer to a
problem, and let the computer deduce the logic behind it. On the other
hand, *unsupervised learning* is a method for finding patterns and
relationship in data sets without any prior knowledge of the system.
Some authours also operate with a third category, namely
*reinforcement learning*. This is a paradigm of learning inspired by
behavioral psychology, where learning is achieved by trial-and-error,
solely from rewards and punishment.
Another way to categorize machine learning tasks is to consider the
desired output of a system. Some of the most common tasks are:
* Classification: Outputs are divided into two or more classes. The goal is to produce a model that assigns inputs into one of these classes. An example is to identify digits based on pictures of hand-written ones. Classification is typically supervised learning.
* Regression: Finding a functional relationship between an input data set and a reference data set. The goal is to construct a function that maps input data to continuous output values.
* Clustering: Data are divided into groups with certain common traits, without knowing the different groups beforehand. It is thus a form of unsupervised learning.
The methods we cover have three main topics in common, irrespective of
whether we deal with supervised or unsupervised learning. The first
ingredient is normally our data set (which can be subdivided into
training and test data), the second item is a model which is normally
a function of some parameters. The model reflects our knowledge of
the system (or lack thereof). As an example, if we know that our data
show a behavior similar to what would be predicted by a polynomial,
fitting our data to a polynomial of some degree would then determin
our model.
The last ingredient is a so-called **cost**
function which allows us to present an estimate on how good our model
is in reproducing the data it is supposed to train.
Here we will build our machine learning approach on elements of the
statistical foundation discussed above, with elements from data
analysis, stochastic processes etc. We will discuss the following
machine learning algorithms
1. Linear regression and its variants
2. Decision tree algorithms, from single trees to random forests
3. Bayesian statistics and regression
4. Support vector machines and finally various variants of
5. Artifical neural networks and deep learning, including convolutional neural networks and Bayesian neural networks
6. Networks for unsupervised learning using for example reduced Boltzmann machines.
## Choice of programming language
Python plays nowadays a central role in the development of machine
learning techniques and tools for data analysis. In particular, seen
the wealth of machine learning and data analysis libraries written in
Python, easy to use libraries with immediate visualization(and not the
least impressive galleries of existing examples), the popularity of the
Jupyter notebook framework with the possibility to run **R** codes or
compiled programs written in C++, and much more made our choice of
programming language for this series of lectures easy. However,
since the focus here is not only on using existing Python libraries such
as **Scikit-Learn** or **Tensorflow**, but also on developing your own
algorithms and codes, we will as far as possible present many of these
algorithms either as a Python codes or C++ or Fortran (or other languages) codes.
The reason we also focus on compiled languages like C++ (or
Fortran), is that Python is still notoriously slow when we do not
utilize highly streamlined computational libraries like
[Lapack](http://www.netlib.org/lapack/) or other numerical libraries
written in compiled languages (many of these libraries are written in
Fortran). Although a project like [Numba](https://numba.pydata.org/)
holds great promise for speeding up the unrolling of lengthy loops, C++
and Fortran are presently still the performance winners. Numba gives
you potentially the power to speed up your applications with high
performance functions written directly in Python. In particular,
array-oriented and math-heavy Python code can achieve similar
performance to C, C++ and Fortran. However, even with these speed-ups,
for codes involving heavy Markov Chain Monte Carlo analyses and
optimizations of cost functions, C++/C or Fortran codes tend to
outperform Python codes.
Presently thus, the community tends to let
code written in C++/C or Fortran do the heavy duty numerical
number crunching and leave the post-analysis of the data to the above
mentioned Python modules or software packages. However, with the developments taking place in for example the Python community, and seen
the changes during the last decade, the above situation may change swiftly in the not too distant future.
Many of the examples we discuss in this series of lectures come with
existing data files or provide code examples which produce the data to
be analyzed. Most of the applications we will discuss deal with
small data sets (less than a terabyte of information) and can easily
be analyzed and tested on standard off the shelf laptops you find in general
stores.
## Data handling, machine learning and ethical aspects
In most of the cases we will study, we will either generate the data
to analyze ourselves (both for supervised learning and unsupervised
learning) or we will recur again and again to data present in say
**Scikit-Learn** or **Tensorflow**. Many of the examples we end up
dealing with are from a privacy and data protection point of view,
rather inoccuous and boring results of numerical
calculations. However, this does not hinder us from developing a sound
ethical attitude to the data we use, how we analyze the data and how
we handle the data.
The most immediate and simplest possible ethical aspects deal with our
approach to the scientific process. Nowadays, with version control
software like [Git](https://git-scm.com/) and various online
repositories like [Github](https://github.com/),
[Gitlab](https://about.gitlab.com/) etc, we can easily make our codes
and data sets we have used, freely and easily accessible to a wider
community. This helps us almost automagically in making our science
reproducible. The large open-source development communities involved
in say [Scikit-Learn](http://scikit-learn.org/stable/),
[Tensorflow](https://www.tensorflow.org/),
[PyTorch](http://pytorch.org/) and [Keras](https://keras.io/), are
all excellent examples of this. The codes can be tested and improved
upon continuosly, helping thereby our scientific community at large in
developing data analysis and machine learning tools. It is much
easier today to gain traction and acceptance for making your science
reproducible. From a societal stand, this is an important element
since many of the developers are employees of large public institutions like
universities and research labs. Our fellow taxpayers do deserve to get
something back for their bucks.
However, this more mechanical aspect of the ethics of science (in
particular the reproducibility of scientific results) is something
which is obvious and everybody should do so as part of the dialectics of
science. The fact that many scientists are not willing to share their codes or
data is detrimental to the scientific discourse.
Before we proceed, we should add a disclaimer. Even though
we may dream of computers developing some kind of higher learning
capabilities, at the end (even if the artificial intelligence
community keeps touting our ears full of fancy futuristic avenues), it is we, yes you reading these lines,
who end up constructing and instructing, via various algorithms, the
machine learning approaches. Self-driving cars for example, rely on sofisticated
programs which take into account all possible situations a car can
encounter. In addition, extensive usage of training data from GPS
information, maps etc, are typically fed into the software for
self-driving cars. Adding to this various sensors and cameras that
feed information to the programs, there are zillions of ethical issues
which arise from this.
For self-driving cars, where basically many of the standard machine
learning algorithms discussed here enter into the codes, at a certain
stage we have to make choices. Yes, we , the lads and lasses who wrote
a program for a specific brand of a self-driving car. As an example,
all carmakers have as their utmost priority the security of the
driver and the accompanying passengers. A famous European carmaker, which is
one of the leaders in the market of self-driving cars, had **if**
statements of the following type: suppose there are two obstacles in
front of you and you cannot avoid to collide with one of them. One of
the obstacles is a monstertruck while the other one is a kindergarten
class trying to cross the road. The self-driving car algo would then
opt for the hitting the small folks instead of the monstertruck, since
the likelihood of surving a collision with our future citizens, is
much higher.
This leads to serious ethical aspects. Why should we opt for such an
option? Who decides and who is entitled to make such choices? Keep in
mind that many of the algorithms you will encounter in this series of
lectures or hear about later, are indeed based on simple programming
instructions. And you are very likely to be one of the people who may
end up writing such a code. Thus, developing a sound ethical attitude
to what we do, an approach well beyond the simple mechanistic one of
making our science available and reproducible, is much needed. The
example of the self-driving cars is just one of infinitely many cases
where we have to make choices. When you analyze data on economic
inequalities, who guarantees that you are not weighting some data in a
particular way, perhaps because you dearly want a specific conclusion
which may support your political views? Or what about the recent
claims that a famous IT company like Apple has a sexist bias on the
their recently [launched credit card](https://qz.com/1748321/the-role-of-goldman-sachs-algorithms-in-the-apple-credit-card-scandal/)?
We do not have the answers here, nor will we venture into a deeper
discussions of these aspects, but we want you think over these topics
in a more overarching way. A statistical data analysis with its dry
numbers and graphs meant to guide the eye, does not necessarily
reflect the truth, whatever that is. As a scientist, and after a
university education, you are supposedly a better citizen, with an
improved critical view and understanding of the scientific method, and
perhaps some deeper understanding of the ethics of science at
large. Use these insights. Be a critical citizen. You owe it to our
society.
+123
View File
@@ -0,0 +1,123 @@
import tensorflow.compat.v1 as tf
tf.disable_v2_behavior()
#tf.reset_default_graph()
# import necessary packages
import numpy as np
import matplotlib.pyplot as plt
from sklearn import datasets
# ensure the same random numbers appear every time
np.random.seed(0)
plt.rcParams['figure.figsize'] = (12,12)
# download MNIST dataset
digits = datasets.load_digits()
# define inputs and labels
inputs = digits.images
labels = digits.target
print("inputs = (n_inputs, pixel_width, pixel_height) = " + str(inputs.shape))
print("labels = (n_inputs) = " + str(labels.shape))
# flatten the image
# the value -1 means dimension is inferred from the remaining dimensions: 8x8 = 64
n_inputs = len(inputs)
inputs = inputs.reshape(n_inputs, -1)
print("X = (n_inputs, n_features) = " + str(inputs.shape))
# choose some random images to display
indices = np.arange(n_inputs)
random_indices = np.random.choice(indices, size=5)
for i, image in enumerate(digits.images[random_indices]):
plt.subplot(1, 5, i+1)
plt.axis('off')
plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
plt.title("Label: %d" % digits.target[random_indices[i]])
plt.show()
from keras.utils import to_categorical
from sklearn.model_selection import train_test_split
# one-hot representation of labels
labels = to_categorical(labels)
# split into train and test data
train_size = 0.8
test_size = 1 - train_size
X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
test_size=test_size)
epochs = 100
batch_size = 100
n_neurons_layer1 = 100
n_neurons_layer2 = 50
n_categories = 10
eta_vals = np.logspace(-5, 1, 7)
lmbd_vals = np.logspace(-5, 1, 7)
from keras.models import Sequential
from keras.layers import Dense
from keras.regularizers import l2
from keras.optimizers import SGD
def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
model = Sequential()
model.add(Dense(n_neurons_layer1, activation='sigmoid', kernel_regularizer=l2(lmbd)))
model.add(Dense(n_neurons_layer2, activation='sigmoid', kernel_regularizer=l2(lmbd)))
model.add(Dense(n_categories, activation='softmax'))
sgd = SGD(lr=eta)
model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
return model
DNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
for i, eta in enumerate(eta_vals):
for j, lmbd in enumerate(lmbd_vals):
DNN = create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories,eta=eta, lmbd=lmbd)
DNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
scores = DNN.evaluate(X_test, Y_test)
DNN_keras[i][j] = DNN
print("Learning rate = ", eta)
print("Lambda = ", lmbd)
print("Test accuracy: %.3f" % scores[1])
print()
import seaborn as sns
sns.set()
train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
for i in range(len(eta_vals)):
for j in range(len(lmbd_vals)):
DNN = DNN_keras[i][j]
train_accuracy[i][j] = DNN.evaluate(X_train, Y_train)[1]
test_accuracy[i][j] = DNN.evaluate(X_test, Y_test)[1]
fig, ax = plt.subplots(figsize = (10, 10))
sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
ax.set_title("Training Accuracy")
ax.set_ylabel("$\eta$")
ax.set_xlabel("$\lambda$")
plt.show()
fig, ax = plt.subplots(figsize = (10, 10))
sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
ax.set_title("Test Accuracy")
ax.set_ylabel("$\eta$")
ax.set_xlabel("$\lambda$")
plt.show()
+42
View File
@@ -0,0 +1,42 @@
from numpy import *
from numpy.random import randint, randn
from time import time
import matplotlib.mlab as mlab
import matplotlib.pyplot as plt
# Returns mean of bootstrap samples
def stat(data):
return mean(data)
# Bootstrap algorithm
def bootstrap(data, statistic, R):
t = zeros(R); n = len(data); inds = arange(n); t0 = time()
# non-parametric bootstrap
for i in range(R):
t[i] = statistic(data[randint(0,n,n)])
# analysis
print("Runtime: %g sec" % (time()-t0)); print("Bootstrap Statistics :")
print("original bias std. error")
print("%8g %8g %14g %15g" % (statistic(data), std(data),mean(t),std(t)))
return t
mu, sigma = 100, 15
datapoints = 10000
x = mu + sigma*random.randn(datapoints)
# bootstrap returns the data sample
t = bootstrap(x, stat, datapoints)
# the histogram of the bootstrapped data
n, binsboot, patches = plt.hist(t, 50, facecolor='red', alpha=0.75)
# add a 'best fit' line
#y = mlab.normpdf( binsboot, mean(t), std(t))
#lt = plt.plot(binsboot, y, 'r--', linewidth=1)
plt.xlabel('Smarts')
plt.ylabel('Probability')
plt.axis([95, 105, 0, 10.0])
plt.grid(True)
plt.show()