From cbb47a8053684ab2975ad848270a34bbfb183399 Mon Sep 17 00:00:00 2001 From: Morten Hjorth-Jensen Date: Sun, 20 Aug 2023 21:16:10 +0200 Subject: [PATCH] update --- doc/pub/week34/html/._week34-bs000.html | 100 +- doc/pub/week34/html/._week34-bs001.html | 102 +- doc/pub/week34/html/._week34-bs002.html | 100 +- doc/pub/week34/html/._week34-bs003.html | 100 +- doc/pub/week34/html/._week34-bs004.html | 100 +- doc/pub/week34/html/._week34-bs005.html | 100 +- doc/pub/week34/html/._week34-bs006.html | 100 +- doc/pub/week34/html/._week34-bs007.html | 100 +- doc/pub/week34/html/._week34-bs008.html | 100 +- doc/pub/week34/html/._week34-bs009.html | 100 +- doc/pub/week34/html/._week34-bs010.html | 100 +- doc/pub/week34/html/._week34-bs011.html | 100 +- doc/pub/week34/html/._week34-bs012.html | 100 +- doc/pub/week34/html/._week34-bs013.html | 100 +- doc/pub/week34/html/._week34-bs014.html | 100 +- doc/pub/week34/html/._week34-bs015.html | 100 +- doc/pub/week34/html/._week34-bs016.html | 104 +- doc/pub/week34/html/._week34-bs017.html | 100 +- doc/pub/week34/html/._week34-bs018.html | 100 +- doc/pub/week34/html/._week34-bs019.html | 100 +- doc/pub/week34/html/._week34-bs020.html | 100 +- doc/pub/week34/html/._week34-bs021.html | 100 +- doc/pub/week34/html/._week34-bs022.html | 100 +- doc/pub/week34/html/._week34-bs023.html | 100 +- doc/pub/week34/html/._week34-bs024.html | 100 +- doc/pub/week34/html/._week34-bs025.html | 100 +- doc/pub/week34/html/._week34-bs026.html | 100 +- doc/pub/week34/html/._week34-bs027.html | 104 +- doc/pub/week34/html/._week34-bs028.html | 105 +- doc/pub/week34/html/._week34-bs029.html | 100 +- doc/pub/week34/html/._week34-bs030.html | 361 ++- doc/pub/week34/html/._week34-bs031.html | 382 ++- doc/pub/week34/html/._week34-bs032.html | 355 ++- doc/pub/week34/html/._week34-bs033.html | 799 +++++- doc/pub/week34/html/._week34-bs034.html | 383 +-- doc/pub/week34/html/._week34-bs035.html | 354 +-- doc/pub/week34/html/._week34-bs036.html | 128 +- doc/pub/week34/html/._week34-bs037.html | 1010 +------- doc/pub/week34/html/._week34-bs038.html | 130 +- doc/pub/week34/html/._week34-bs039.html | 124 +- doc/pub/week34/html/._week34-bs040.html | 149 +- doc/pub/week34/html/._week34-bs041.html | 131 +- doc/pub/week34/html/._week34-bs042.html | 122 +- doc/pub/week34/html/._week34-bs043.html | 119 +- doc/pub/week34/html/._week34-bs044.html | 226 +- doc/pub/week34/html/._week34-bs045.html | 138 +- doc/pub/week34/html/._week34-bs046.html | 151 +- doc/pub/week34/html/._week34-bs047.html | 145 +- doc/pub/week34/html/._week34-bs048.html | 203 +- doc/pub/week34/html/._week34-bs049.html | 128 +- doc/pub/week34/html/._week34-bs050.html | 229 +- doc/pub/week34/html/._week34-bs051.html | 230 +- doc/pub/week34/html/._week34-bs052.html | 134 +- doc/pub/week34/html/._week34-bs053.html | 120 +- doc/pub/week34/html/._week34-bs054.html | 205 +- doc/pub/week34/html/._week34-bs055.html | 227 +- doc/pub/week34/html/._week34-bs056.html | 130 +- doc/pub/week34/html/._week34-bs057.html | 143 +- doc/pub/week34/html/._week34-bs058.html | 139 +- doc/pub/week34/html/._week34-bs059.html | 232 +- doc/pub/week34/html/._week34-bs060.html | 213 +- doc/pub/week34/html/._week34-bs061.html | 343 ++- doc/pub/week34/html/week34-bs.html | 100 +- doc/pub/week34/html/week34-reveal.html | 440 +--- doc/pub/week34/html/week34-solarized.html | 425 +--- doc/pub/week34/html/week34.html | 425 +--- doc/pub/week34/ipynb/ipynb-week34-src.tar.gz | Bin 103516 -> 103516 bytes doc/pub/week34/ipynb/week34.ipynb | 2353 ++++++++++-------- doc/src/week34/week34.do.txt | 9 +- 69 files changed, 6436 insertions(+), 8214 deletions(-) diff --git a/doc/pub/week34/html/._week34-bs000.html b/doc/pub/week34/html/._week34-bs000.html index 9aaab6cf1..19b4a2d84 100644 --- a/doc/pub/week34/html/._week34-bs000.html +++ b/doc/pub/week34/html/._week34-bs000.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
  • Installing R, C++, cython or Julia
  • Installing R, C++, cython, Numba etc
  • Numpy examples and Important Matrix and vector handling packages
  • -
  • Basic Matrix Features
  • -
  •    Some famous Matrices
  • -
  •    More Basic Matrix Features
  • -
  • Numpy and arrays
  • -
  • Matrices in Python
  • -
  • Meet the Pandas
  • -
  • Friday August 27
  • -
  •    Simple linear regression model using scikit-learn
  • -
  •    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
  • -
  •    Organizing our data
  • -
  •    Seeing the wood for the trees
  • -
  •    And what about using neural networks?
  • -
  • A first summary
  • -
  • Why Linear Regression (aka Ordinary Least Squares and family)
  • -
  • Regression analysis, overarching aims
  • -
  • Regression analysis, overarching aims II
  • -
  • Examples
  • -
  • General linear models
  • -
  • Rewriting the fitting procedure as a linear algebra problem
  • -
  • Rewriting the fitting procedure as a linear algebra problem, more details
  • -
  • Generalizing the fitting procedure as a linear algebra problem
  • -
  • Generalizing the fitting procedure as a linear algebra problem
  • -
  • Optimizing our parameters
  • -
  • Our model for the nuclear binding energies
  • -
  • Optimizing our parameters, more details
  • -
  • Interpretations and optimizing our parameters
  • -
  • Interpretations and optimizing our parameters
  • -
  • Some useful matrix and vector expressions
  • -
  • Interpretations and optimizing our parameters
  • -
  • Own code for Ordinary Least Squares
  • -
  • Adding error analysis and training set up
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • Fitting an Equation of State for Dense Nuclear Matter
  • -
  • The code
  • -
  • Splitting our Data in Training and Test data
  • -
  • Exercises
  • -
  • Exercise 1: Setting up various Python environments
  • -
  • Exercise 2: making your own data and exploring scikit-learn
  • -
  • Exercise 3: Normalizing our data
  • +
  • Numpy and arrays
  • +
  • Matrices in Python
  • +
  • Meet the Pandas
  • +
  •    Simple linear regression model using scikit-learn
  • +
  •    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
  • +
  •    Organizing our data
  • +
  •    And what about using neural networks?
  • +
  • A first summary
  • +
  • Why Linear Regression (aka Ordinary Least Squares and family)
  • +
  • Regression analysis, overarching aims
  • +
  • Regression analysis, overarching aims II
  • +
  • Examples
  • +
  • General linear models
  • +
  • Rewriting the fitting procedure as a linear algebra problem
  • +
  • Rewriting the fitting procedure as a linear algebra problem, more details
  • +
  • Generalizing the fitting procedure as a linear algebra problem
  • +
  • Generalizing the fitting procedure as a linear algebra problem
  • +
  • Optimizing our parameters
  • +
  • Our model for the nuclear binding energies
  • +
  • Optimizing our parameters, more details
  • +
  • Interpretations and optimizing our parameters
  • +
  • Interpretations and optimizing our parameters
  • +
  • Some useful matrix and vector expressions
  • +
  • Interpretations and optimizing our parameters
  • +
  • Own code for Ordinary Least Squares
  • +
  • Adding error analysis and training set up
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • Fitting an Equation of State for Dense Nuclear Matter
  • +
  • The code
  • +
  • Splitting our Data in Training and Test data
  • +
  • Exercises
  • +
  • Exercise 1: Setting up various Python environments
  • +
  • Exercise 2: making your own data and exploring scikit-learn
  • +
  • Exercise 3: Split data in test and training data
  • @@ -394,7 +378,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 66
  • +
  • 62
  • »
  • diff --git a/doc/pub/week34/html/._week34-bs001.html b/doc/pub/week34/html/._week34-bs001.html index 6f10f533d..4059ee806 100644 --- a/doc/pub/week34/html/._week34-bs001.html +++ b/doc/pub/week34/html/._week34-bs001.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
  • Installing R, C++, cython or Julia
  • Installing R, C++, cython, Numba etc
  • Numpy examples and Important Matrix and vector handling packages
  • -
  • Basic Matrix Features
  • -
  •    Some famous Matrices
  • -
  •    More Basic Matrix Features
  • -
  • Numpy and arrays
  • -
  • Matrices in Python
  • -
  • Meet the Pandas
  • -
  • Friday August 27
  • -
  •    Simple linear regression model using scikit-learn
  • -
  •    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
  • -
  •    Organizing our data
  • -
  •    Seeing the wood for the trees
  • -
  •    And what about using neural networks?
  • -
  • A first summary
  • -
  • Why Linear Regression (aka Ordinary Least Squares and family)
  • -
  • Regression analysis, overarching aims
  • -
  • Regression analysis, overarching aims II
  • -
  • Examples
  • -
  • General linear models
  • -
  • Rewriting the fitting procedure as a linear algebra problem
  • -
  • Rewriting the fitting procedure as a linear algebra problem, more details
  • -
  • Generalizing the fitting procedure as a linear algebra problem
  • -
  • Generalizing the fitting procedure as a linear algebra problem
  • -
  • Optimizing our parameters
  • -
  • Our model for the nuclear binding energies
  • -
  • Optimizing our parameters, more details
  • -
  • Interpretations and optimizing our parameters
  • -
  • Interpretations and optimizing our parameters
  • -
  • Some useful matrix and vector expressions
  • -
  • Interpretations and optimizing our parameters
  • -
  • Own code for Ordinary Least Squares
  • -
  • Adding error analysis and training set up
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • The \( \chi^2 \) function
  • -
  • Fitting an Equation of State for Dense Nuclear Matter
  • -
  • The code
  • -
  • Splitting our Data in Training and Test data
  • -
  • Exercises
  • -
  • Exercise 1: Setting up various Python environments
  • -
  • Exercise 2: making your own data and exploring scikit-learn
  • -
  • Exercise 3: Normalizing our data
  • +
  • Numpy and arrays
  • +
  • Matrices in Python
  • +
  • Meet the Pandas
  • +
  •    Simple linear regression model using scikit-learn
  • +
  •    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
  • +
  •    Organizing our data
  • +
  •    And what about using neural networks?
  • +
  • A first summary
  • +
  • Why Linear Regression (aka Ordinary Least Squares and family)
  • +
  • Regression analysis, overarching aims
  • +
  • Regression analysis, overarching aims II
  • +
  • Examples
  • +
  • General linear models
  • +
  • Rewriting the fitting procedure as a linear algebra problem
  • +
  • Rewriting the fitting procedure as a linear algebra problem, more details
  • +
  • Generalizing the fitting procedure as a linear algebra problem
  • +
  • Generalizing the fitting procedure as a linear algebra problem
  • +
  • Optimizing our parameters
  • +
  • Our model for the nuclear binding energies
  • +
  • Optimizing our parameters, more details
  • +
  • Interpretations and optimizing our parameters
  • +
  • Interpretations and optimizing our parameters
  • +
  • Some useful matrix and vector expressions
  • +
  • Interpretations and optimizing our parameters
  • +
  • Own code for Ordinary Least Squares
  • +
  • Adding error analysis and training set up
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • The \( \chi^2 \) function
  • +
  • Fitting an Equation of State for Dense Nuclear Matter
  • +
  • The code
  • +
  • Splitting our Data in Training and Test data
  • +
  • Exercises
  • +
  • Exercise 1: Setting up various Python environments
  • +
  • Exercise 2: making your own data and exploring scikit-learn
  • +
  • Exercise 3: Split data in test and training data
  • @@ -355,7 +339,7 @@ MathJax.Hub.Config({
    1. The sessions on Tuesdays and Wednesdays last four hours for each group (four groups in total) and will include lectures in a flipped mode (promoting active learning) and work on exercices and projects.
    2. -
    3. The sessions will begin with lectures, discussions, questions and answers about the material to be covered every week.
    4. +
    5. The sessions will begin with lectures, discussions, questions and answers about the material to be covered every week. Videos and teaching material will be announced in due time.
    6. There are four groups:
    7. diff --git a/doc/pub/week34/html/._week34-bs002.html b/doc/pub/week34/html/._week34-bs002.html index b3b507791..b3f6f153d 100644 --- a/doc/pub/week34/html/._week34-bs002.html +++ b/doc/pub/week34/html/._week34-bs002.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8. Installing R, C++, cython or Julia
    9. Installing R, C++, cython, Numba etc
    10. Numpy examples and Important Matrix and vector handling packages
    11. -
    12. Basic Matrix Features
    13. -
    14.    Some famous Matrices
    15. -
    16.    More Basic Matrix Features
    17. -
    18. Numpy and arrays
    19. -
    20. Matrices in Python
    21. -
    22. Meet the Pandas
    23. -
    24. Friday August 27
    25. -
    26.    Simple linear regression model using scikit-learn
    27. -
    28.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    29. -
    30.    Organizing our data
    31. -
    32.    Seeing the wood for the trees
    33. -
    34.    And what about using neural networks?
    35. -
    36. A first summary
    37. -
    38. Why Linear Regression (aka Ordinary Least Squares and family)
    39. -
    40. Regression analysis, overarching aims
    41. -
    42. Regression analysis, overarching aims II
    43. -
    44. Examples
    45. -
    46. General linear models
    47. -
    48. Rewriting the fitting procedure as a linear algebra problem
    49. -
    50. Rewriting the fitting procedure as a linear algebra problem, more details
    51. -
    52. Generalizing the fitting procedure as a linear algebra problem
    53. -
    54. Generalizing the fitting procedure as a linear algebra problem
    55. -
    56. Optimizing our parameters
    57. -
    58. Our model for the nuclear binding energies
    59. -
    60. Optimizing our parameters, more details
    61. -
    62. Interpretations and optimizing our parameters
    63. -
    64. Interpretations and optimizing our parameters
    65. -
    66. Some useful matrix and vector expressions
    67. -
    68. Interpretations and optimizing our parameters
    69. -
    70. Own code for Ordinary Least Squares
    71. -
    72. Adding error analysis and training set up
    73. -
    74. The \( \chi^2 \) function
    75. -
    76. The \( \chi^2 \) function
    77. -
    78. The \( \chi^2 \) function
    79. -
    80. The \( \chi^2 \) function
    81. -
    82. The \( \chi^2 \) function
    83. -
    84. The \( \chi^2 \) function
    85. -
    86. Fitting an Equation of State for Dense Nuclear Matter
    87. -
    88. The code
    89. -
    90. Splitting our Data in Training and Test data
    91. -
    92. Exercises
    93. -
    94. Exercise 1: Setting up various Python environments
    95. -
    96. Exercise 2: making your own data and exploring scikit-learn
    97. -
    98. Exercise 3: Normalizing our data
    99. +
    100. Numpy and arrays
    101. +
    102. Matrices in Python
    103. +
    104. Meet the Pandas
    105. +
    106.    Simple linear regression model using scikit-learn
    107. +
    108.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    109. +
    110.    Organizing our data
    111. +
    112.    And what about using neural networks?
    113. +
    114. A first summary
    115. +
    116. Why Linear Regression (aka Ordinary Least Squares and family)
    117. +
    118. Regression analysis, overarching aims
    119. +
    120. Regression analysis, overarching aims II
    121. +
    122. Examples
    123. +
    124. General linear models
    125. +
    126. Rewriting the fitting procedure as a linear algebra problem
    127. +
    128. Rewriting the fitting procedure as a linear algebra problem, more details
    129. +
    130. Generalizing the fitting procedure as a linear algebra problem
    131. +
    132. Generalizing the fitting procedure as a linear algebra problem
    133. +
    134. Optimizing our parameters
    135. +
    136. Our model for the nuclear binding energies
    137. +
    138. Optimizing our parameters, more details
    139. +
    140. Interpretations and optimizing our parameters
    141. +
    142. Interpretations and optimizing our parameters
    143. +
    144. Some useful matrix and vector expressions
    145. +
    146. Interpretations and optimizing our parameters
    147. +
    148. Own code for Ordinary Least Squares
    149. +
    150. Adding error analysis and training set up
    151. +
    152. The \( \chi^2 \) function
    153. +
    154. The \( \chi^2 \) function
    155. +
    156. The \( \chi^2 \) function
    157. +
    158. The \( \chi^2 \) function
    159. +
    160. The \( \chi^2 \) function
    161. +
    162. The \( \chi^2 \) function
    163. +
    164. Fitting an Equation of State for Dense Nuclear Matter
    165. +
    166. The code
    167. +
    168. Splitting our Data in Training and Test data
    169. +
    170. Exercises
    171. +
    172. Exercise 1: Setting up various Python environments
    173. +
    174. Exercise 2: making your own data and exploring scikit-learn
    175. +
    176. Exercise 3: Split data in test and training data
    177. @@ -382,7 +366,7 @@ MathJax.Hub.Config({
    178. 11
    179. 12
    180. ...
    181. -
    182. 66
    183. +
    184. 62
    185. »
    186. diff --git a/doc/pub/week34/html/._week34-bs003.html b/doc/pub/week34/html/._week34-bs003.html index a5ea46b15..2c7bc9acd 100644 --- a/doc/pub/week34/html/._week34-bs003.html +++ b/doc/pub/week34/html/._week34-bs003.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    187. Installing R, C++, cython or Julia
    188. Installing R, C++, cython, Numba etc
    189. Numpy examples and Important Matrix and vector handling packages
    190. -
    191. Basic Matrix Features
    192. -
    193.    Some famous Matrices
    194. -
    195.    More Basic Matrix Features
    196. -
    197. Numpy and arrays
    198. -
    199. Matrices in Python
    200. -
    201. Meet the Pandas
    202. -
    203. Friday August 27
    204. -
    205.    Simple linear regression model using scikit-learn
    206. -
    207.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    208. -
    209.    Organizing our data
    210. -
    211.    Seeing the wood for the trees
    212. -
    213.    And what about using neural networks?
    214. -
    215. A first summary
    216. -
    217. Why Linear Regression (aka Ordinary Least Squares and family)
    218. -
    219. Regression analysis, overarching aims
    220. -
    221. Regression analysis, overarching aims II
    222. -
    223. Examples
    224. -
    225. General linear models
    226. -
    227. Rewriting the fitting procedure as a linear algebra problem
    228. -
    229. Rewriting the fitting procedure as a linear algebra problem, more details
    230. -
    231. Generalizing the fitting procedure as a linear algebra problem
    232. -
    233. Generalizing the fitting procedure as a linear algebra problem
    234. -
    235. Optimizing our parameters
    236. -
    237. Our model for the nuclear binding energies
    238. -
    239. Optimizing our parameters, more details
    240. -
    241. Interpretations and optimizing our parameters
    242. -
    243. Interpretations and optimizing our parameters
    244. -
    245. Some useful matrix and vector expressions
    246. -
    247. Interpretations and optimizing our parameters
    248. -
    249. Own code for Ordinary Least Squares
    250. -
    251. Adding error analysis and training set up
    252. -
    253. The \( \chi^2 \) function
    254. -
    255. The \( \chi^2 \) function
    256. -
    257. The \( \chi^2 \) function
    258. -
    259. The \( \chi^2 \) function
    260. -
    261. The \( \chi^2 \) function
    262. -
    263. The \( \chi^2 \) function
    264. -
    265. Fitting an Equation of State for Dense Nuclear Matter
    266. -
    267. The code
    268. -
    269. Splitting our Data in Training and Test data
    270. -
    271. Exercises
    272. -
    273. Exercise 1: Setting up various Python environments
    274. -
    275. Exercise 2: making your own data and exploring scikit-learn
    276. -
    277. Exercise 3: Normalizing our data
    278. +
    279. Numpy and arrays
    280. +
    281. Matrices in Python
    282. +
    283. Meet the Pandas
    284. +
    285.    Simple linear regression model using scikit-learn
    286. +
    287.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    288. +
    289.    Organizing our data
    290. +
    291.    And what about using neural networks?
    292. +
    293. A first summary
    294. +
    295. Why Linear Regression (aka Ordinary Least Squares and family)
    296. +
    297. Regression analysis, overarching aims
    298. +
    299. Regression analysis, overarching aims II
    300. +
    301. Examples
    302. +
    303. General linear models
    304. +
    305. Rewriting the fitting procedure as a linear algebra problem
    306. +
    307. Rewriting the fitting procedure as a linear algebra problem, more details
    308. +
    309. Generalizing the fitting procedure as a linear algebra problem
    310. +
    311. Generalizing the fitting procedure as a linear algebra problem
    312. +
    313. Optimizing our parameters
    314. +
    315. Our model for the nuclear binding energies
    316. +
    317. Optimizing our parameters, more details
    318. +
    319. Interpretations and optimizing our parameters
    320. +
    321. Interpretations and optimizing our parameters
    322. +
    323. Some useful matrix and vector expressions
    324. +
    325. Interpretations and optimizing our parameters
    326. +
    327. Own code for Ordinary Least Squares
    328. +
    329. Adding error analysis and training set up
    330. +
    331. The \( \chi^2 \) function
    332. +
    333. The \( \chi^2 \) function
    334. +
    335. The \( \chi^2 \) function
    336. +
    337. The \( \chi^2 \) function
    338. +
    339. The \( \chi^2 \) function
    340. +
    341. The \( \chi^2 \) function
    342. +
    343. Fitting an Equation of State for Dense Nuclear Matter
    344. +
    345. The code
    346. +
    347. Splitting our Data in Training and Test data
    348. +
    349. Exercises
    350. +
    351. Exercise 1: Setting up various Python environments
    352. +
    353. Exercise 2: making your own data and exploring scikit-learn
    354. +
    355. Exercise 3: Split data in test and training data
    356. @@ -387,7 +371,7 @@ MathJax.Hub.Config({
    357. 12
    358. 13
    359. ...
    360. -
    361. 66
    362. +
    363. 62
    364. »
    365. diff --git a/doc/pub/week34/html/._week34-bs004.html b/doc/pub/week34/html/._week34-bs004.html index 8e35c8c24..3ce4433bd 100644 --- a/doc/pub/week34/html/._week34-bs004.html +++ b/doc/pub/week34/html/._week34-bs004.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    366. Installing R, C++, cython or Julia
    367. Installing R, C++, cython, Numba etc
    368. Numpy examples and Important Matrix and vector handling packages
    369. -
    370. Basic Matrix Features
    371. -
    372.    Some famous Matrices
    373. -
    374.    More Basic Matrix Features
    375. -
    376. Numpy and arrays
    377. -
    378. Matrices in Python
    379. -
    380. Meet the Pandas
    381. -
    382. Friday August 27
    383. -
    384.    Simple linear regression model using scikit-learn
    385. -
    386.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    387. -
    388.    Organizing our data
    389. -
    390.    Seeing the wood for the trees
    391. -
    392.    And what about using neural networks?
    393. -
    394. A first summary
    395. -
    396. Why Linear Regression (aka Ordinary Least Squares and family)
    397. -
    398. Regression analysis, overarching aims
    399. -
    400. Regression analysis, overarching aims II
    401. -
    402. Examples
    403. -
    404. General linear models
    405. -
    406. Rewriting the fitting procedure as a linear algebra problem
    407. -
    408. Rewriting the fitting procedure as a linear algebra problem, more details
    409. -
    410. Generalizing the fitting procedure as a linear algebra problem
    411. -
    412. Generalizing the fitting procedure as a linear algebra problem
    413. -
    414. Optimizing our parameters
    415. -
    416. Our model for the nuclear binding energies
    417. -
    418. Optimizing our parameters, more details
    419. -
    420. Interpretations and optimizing our parameters
    421. -
    422. Interpretations and optimizing our parameters
    423. -
    424. Some useful matrix and vector expressions
    425. -
    426. Interpretations and optimizing our parameters
    427. -
    428. Own code for Ordinary Least Squares
    429. -
    430. Adding error analysis and training set up
    431. -
    432. The \( \chi^2 \) function
    433. -
    434. The \( \chi^2 \) function
    435. -
    436. The \( \chi^2 \) function
    437. -
    438. The \( \chi^2 \) function
    439. -
    440. The \( \chi^2 \) function
    441. -
    442. The \( \chi^2 \) function
    443. -
    444. Fitting an Equation of State for Dense Nuclear Matter
    445. -
    446. The code
    447. -
    448. Splitting our Data in Training and Test data
    449. -
    450. Exercises
    451. -
    452. Exercise 1: Setting up various Python environments
    453. -
    454. Exercise 2: making your own data and exploring scikit-learn
    455. -
    456. Exercise 3: Normalizing our data
    457. +
    458. Numpy and arrays
    459. +
    460. Matrices in Python
    461. +
    462. Meet the Pandas
    463. +
    464.    Simple linear regression model using scikit-learn
    465. +
    466.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    467. +
    468.    Organizing our data
    469. +
    470.    And what about using neural networks?
    471. +
    472. A first summary
    473. +
    474. Why Linear Regression (aka Ordinary Least Squares and family)
    475. +
    476. Regression analysis, overarching aims
    477. +
    478. Regression analysis, overarching aims II
    479. +
    480. Examples
    481. +
    482. General linear models
    483. +
    484. Rewriting the fitting procedure as a linear algebra problem
    485. +
    486. Rewriting the fitting procedure as a linear algebra problem, more details
    487. +
    488. Generalizing the fitting procedure as a linear algebra problem
    489. +
    490. Generalizing the fitting procedure as a linear algebra problem
    491. +
    492. Optimizing our parameters
    493. +
    494. Our model for the nuclear binding energies
    495. +
    496. Optimizing our parameters, more details
    497. +
    498. Interpretations and optimizing our parameters
    499. +
    500. Interpretations and optimizing our parameters
    501. +
    502. Some useful matrix and vector expressions
    503. +
    504. Interpretations and optimizing our parameters
    505. +
    506. Own code for Ordinary Least Squares
    507. +
    508. Adding error analysis and training set up
    509. +
    510. The \( \chi^2 \) function
    511. +
    512. The \( \chi^2 \) function
    513. +
    514. The \( \chi^2 \) function
    515. +
    516. The \( \chi^2 \) function
    517. +
    518. The \( \chi^2 \) function
    519. +
    520. The \( \chi^2 \) function
    521. +
    522. Fitting an Equation of State for Dense Nuclear Matter
    523. +
    524. The code
    525. +
    526. Splitting our Data in Training and Test data
    527. +
    528. Exercises
    529. +
    530. Exercise 1: Setting up various Python environments
    531. +
    532. Exercise 2: making your own data and exploring scikit-learn
    533. +
    534. Exercise 3: Split data in test and training data
    535. @@ -388,7 +372,7 @@ MathJax.Hub.Config({
    536. 13
    537. 14
    538. ...
    539. -
    540. 66
    541. +
    542. 62
    543. »
    544. diff --git a/doc/pub/week34/html/._week34-bs005.html b/doc/pub/week34/html/._week34-bs005.html index e368bfb65..dead6245d 100644 --- a/doc/pub/week34/html/._week34-bs005.html +++ b/doc/pub/week34/html/._week34-bs005.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    545. Installing R, C++, cython or Julia
    546. Installing R, C++, cython, Numba etc
    547. Numpy examples and Important Matrix and vector handling packages
    548. -
    549. Basic Matrix Features
    550. -
    551.    Some famous Matrices
    552. -
    553.    More Basic Matrix Features
    554. -
    555. Numpy and arrays
    556. -
    557. Matrices in Python
    558. -
    559. Meet the Pandas
    560. -
    561. Friday August 27
    562. -
    563.    Simple linear regression model using scikit-learn
    564. -
    565.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    566. -
    567.    Organizing our data
    568. -
    569.    Seeing the wood for the trees
    570. -
    571.    And what about using neural networks?
    572. -
    573. A first summary
    574. -
    575. Why Linear Regression (aka Ordinary Least Squares and family)
    576. -
    577. Regression analysis, overarching aims
    578. -
    579. Regression analysis, overarching aims II
    580. -
    581. Examples
    582. -
    583. General linear models
    584. -
    585. Rewriting the fitting procedure as a linear algebra problem
    586. -
    587. Rewriting the fitting procedure as a linear algebra problem, more details
    588. -
    589. Generalizing the fitting procedure as a linear algebra problem
    590. -
    591. Generalizing the fitting procedure as a linear algebra problem
    592. -
    593. Optimizing our parameters
    594. -
    595. Our model for the nuclear binding energies
    596. -
    597. Optimizing our parameters, more details
    598. -
    599. Interpretations and optimizing our parameters
    600. -
    601. Interpretations and optimizing our parameters
    602. -
    603. Some useful matrix and vector expressions
    604. -
    605. Interpretations and optimizing our parameters
    606. -
    607. Own code for Ordinary Least Squares
    608. -
    609. Adding error analysis and training set up
    610. -
    611. The \( \chi^2 \) function
    612. -
    613. The \( \chi^2 \) function
    614. -
    615. The \( \chi^2 \) function
    616. -
    617. The \( \chi^2 \) function
    618. -
    619. The \( \chi^2 \) function
    620. -
    621. The \( \chi^2 \) function
    622. -
    623. Fitting an Equation of State for Dense Nuclear Matter
    624. -
    625. The code
    626. -
    627. Splitting our Data in Training and Test data
    628. -
    629. Exercises
    630. -
    631. Exercise 1: Setting up various Python environments
    632. -
    633. Exercise 2: making your own data and exploring scikit-learn
    634. -
    635. Exercise 3: Normalizing our data
    636. +
    637. Numpy and arrays
    638. +
    639. Matrices in Python
    640. +
    641. Meet the Pandas
    642. +
    643.    Simple linear regression model using scikit-learn
    644. +
    645.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    646. +
    647.    Organizing our data
    648. +
    649.    And what about using neural networks?
    650. +
    651. A first summary
    652. +
    653. Why Linear Regression (aka Ordinary Least Squares and family)
    654. +
    655. Regression analysis, overarching aims
    656. +
    657. Regression analysis, overarching aims II
    658. +
    659. Examples
    660. +
    661. General linear models
    662. +
    663. Rewriting the fitting procedure as a linear algebra problem
    664. +
    665. Rewriting the fitting procedure as a linear algebra problem, more details
    666. +
    667. Generalizing the fitting procedure as a linear algebra problem
    668. +
    669. Generalizing the fitting procedure as a linear algebra problem
    670. +
    671. Optimizing our parameters
    672. +
    673. Our model for the nuclear binding energies
    674. +
    675. Optimizing our parameters, more details
    676. +
    677. Interpretations and optimizing our parameters
    678. +
    679. Interpretations and optimizing our parameters
    680. +
    681. Some useful matrix and vector expressions
    682. +
    683. Interpretations and optimizing our parameters
    684. +
    685. Own code for Ordinary Least Squares
    686. +
    687. Adding error analysis and training set up
    688. +
    689. The \( \chi^2 \) function
    690. +
    691. The \( \chi^2 \) function
    692. +
    693. The \( \chi^2 \) function
    694. +
    695. The \( \chi^2 \) function
    696. +
    697. The \( \chi^2 \) function
    698. +
    699. The \( \chi^2 \) function
    700. +
    701. Fitting an Equation of State for Dense Nuclear Matter
    702. +
    703. The code
    704. +
    705. Splitting our Data in Training and Test data
    706. +
    707. Exercises
    708. +
    709. Exercise 1: Setting up various Python environments
    710. +
    711. Exercise 2: making your own data and exploring scikit-learn
    712. +
    713. Exercise 3: Split data in test and training data
    714. @@ -377,7 +361,7 @@ MathJax.Hub.Config({
    715. 14
    716. 15
    717. ...
    718. -
    719. 66
    720. +
    721. 62
    722. »
    723. diff --git a/doc/pub/week34/html/._week34-bs006.html b/doc/pub/week34/html/._week34-bs006.html index 6424e0c56..e017e4ba6 100644 --- a/doc/pub/week34/html/._week34-bs006.html +++ b/doc/pub/week34/html/._week34-bs006.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    724. Installing R, C++, cython or Julia
    725. Installing R, C++, cython, Numba etc
    726. Numpy examples and Important Matrix and vector handling packages
    727. -
    728. Basic Matrix Features
    729. -
    730.    Some famous Matrices
    731. -
    732.    More Basic Matrix Features
    733. -
    734. Numpy and arrays
    735. -
    736. Matrices in Python
    737. -
    738. Meet the Pandas
    739. -
    740. Friday August 27
    741. -
    742.    Simple linear regression model using scikit-learn
    743. -
    744.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    745. -
    746.    Organizing our data
    747. -
    748.    Seeing the wood for the trees
    749. -
    750.    And what about using neural networks?
    751. -
    752. A first summary
    753. -
    754. Why Linear Regression (aka Ordinary Least Squares and family)
    755. -
    756. Regression analysis, overarching aims
    757. -
    758. Regression analysis, overarching aims II
    759. -
    760. Examples
    761. -
    762. General linear models
    763. -
    764. Rewriting the fitting procedure as a linear algebra problem
    765. -
    766. Rewriting the fitting procedure as a linear algebra problem, more details
    767. -
    768. Generalizing the fitting procedure as a linear algebra problem
    769. -
    770. Generalizing the fitting procedure as a linear algebra problem
    771. -
    772. Optimizing our parameters
    773. -
    774. Our model for the nuclear binding energies
    775. -
    776. Optimizing our parameters, more details
    777. -
    778. Interpretations and optimizing our parameters
    779. -
    780. Interpretations and optimizing our parameters
    781. -
    782. Some useful matrix and vector expressions
    783. -
    784. Interpretations and optimizing our parameters
    785. -
    786. Own code for Ordinary Least Squares
    787. -
    788. Adding error analysis and training set up
    789. -
    790. The \( \chi^2 \) function
    791. -
    792. The \( \chi^2 \) function
    793. -
    794. The \( \chi^2 \) function
    795. -
    796. The \( \chi^2 \) function
    797. -
    798. The \( \chi^2 \) function
    799. -
    800. The \( \chi^2 \) function
    801. -
    802. Fitting an Equation of State for Dense Nuclear Matter
    803. -
    804. The code
    805. -
    806. Splitting our Data in Training and Test data
    807. -
    808. Exercises
    809. -
    810. Exercise 1: Setting up various Python environments
    811. -
    812. Exercise 2: making your own data and exploring scikit-learn
    813. -
    814. Exercise 3: Normalizing our data
    815. +
    816. Numpy and arrays
    817. +
    818. Matrices in Python
    819. +
    820. Meet the Pandas
    821. +
    822.    Simple linear regression model using scikit-learn
    823. +
    824.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    825. +
    826.    Organizing our data
    827. +
    828.    And what about using neural networks?
    829. +
    830. A first summary
    831. +
    832. Why Linear Regression (aka Ordinary Least Squares and family)
    833. +
    834. Regression analysis, overarching aims
    835. +
    836. Regression analysis, overarching aims II
    837. +
    838. Examples
    839. +
    840. General linear models
    841. +
    842. Rewriting the fitting procedure as a linear algebra problem
    843. +
    844. Rewriting the fitting procedure as a linear algebra problem, more details
    845. +
    846. Generalizing the fitting procedure as a linear algebra problem
    847. +
    848. Generalizing the fitting procedure as a linear algebra problem
    849. +
    850. Optimizing our parameters
    851. +
    852. Our model for the nuclear binding energies
    853. +
    854. Optimizing our parameters, more details
    855. +
    856. Interpretations and optimizing our parameters
    857. +
    858. Interpretations and optimizing our parameters
    859. +
    860. Some useful matrix and vector expressions
    861. +
    862. Interpretations and optimizing our parameters
    863. +
    864. Own code for Ordinary Least Squares
    865. +
    866. Adding error analysis and training set up
    867. +
    868. The \( \chi^2 \) function
    869. +
    870. The \( \chi^2 \) function
    871. +
    872. The \( \chi^2 \) function
    873. +
    874. The \( \chi^2 \) function
    875. +
    876. The \( \chi^2 \) function
    877. +
    878. The \( \chi^2 \) function
    879. +
    880. Fitting an Equation of State for Dense Nuclear Matter
    881. +
    882. The code
    883. +
    884. Splitting our Data in Training and Test data
    885. +
    886. Exercises
    887. +
    888. Exercise 1: Setting up various Python environments
    889. +
    890. Exercise 2: making your own data and exploring scikit-learn
    891. +
    892. Exercise 3: Split data in test and training data
    893. @@ -392,7 +376,7 @@ MathJax.Hub.Config({
    894. 15
    895. 16
    896. ...
    897. -
    898. 66
    899. +
    900. 62
    901. »
    902. diff --git a/doc/pub/week34/html/._week34-bs007.html b/doc/pub/week34/html/._week34-bs007.html index f7b15e20b..480c13442 100644 --- a/doc/pub/week34/html/._week34-bs007.html +++ b/doc/pub/week34/html/._week34-bs007.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    903. Installing R, C++, cython or Julia
    904. Installing R, C++, cython, Numba etc
    905. Numpy examples and Important Matrix and vector handling packages
    906. -
    907. Basic Matrix Features
    908. -
    909.    Some famous Matrices
    910. -
    911.    More Basic Matrix Features
    912. -
    913. Numpy and arrays
    914. -
    915. Matrices in Python
    916. -
    917. Meet the Pandas
    918. -
    919. Friday August 27
    920. -
    921.    Simple linear regression model using scikit-learn
    922. -
    923.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    924. -
    925.    Organizing our data
    926. -
    927.    Seeing the wood for the trees
    928. -
    929.    And what about using neural networks?
    930. -
    931. A first summary
    932. -
    933. Why Linear Regression (aka Ordinary Least Squares and family)
    934. -
    935. Regression analysis, overarching aims
    936. -
    937. Regression analysis, overarching aims II
    938. -
    939. Examples
    940. -
    941. General linear models
    942. -
    943. Rewriting the fitting procedure as a linear algebra problem
    944. -
    945. Rewriting the fitting procedure as a linear algebra problem, more details
    946. -
    947. Generalizing the fitting procedure as a linear algebra problem
    948. -
    949. Generalizing the fitting procedure as a linear algebra problem
    950. -
    951. Optimizing our parameters
    952. -
    953. Our model for the nuclear binding energies
    954. -
    955. Optimizing our parameters, more details
    956. -
    957. Interpretations and optimizing our parameters
    958. -
    959. Interpretations and optimizing our parameters
    960. -
    961. Some useful matrix and vector expressions
    962. -
    963. Interpretations and optimizing our parameters
    964. -
    965. Own code for Ordinary Least Squares
    966. -
    967. Adding error analysis and training set up
    968. -
    969. The \( \chi^2 \) function
    970. -
    971. The \( \chi^2 \) function
    972. -
    973. The \( \chi^2 \) function
    974. -
    975. The \( \chi^2 \) function
    976. -
    977. The \( \chi^2 \) function
    978. -
    979. The \( \chi^2 \) function
    980. -
    981. Fitting an Equation of State for Dense Nuclear Matter
    982. -
    983. The code
    984. -
    985. Splitting our Data in Training and Test data
    986. -
    987. Exercises
    988. -
    989. Exercise 1: Setting up various Python environments
    990. -
    991. Exercise 2: making your own data and exploring scikit-learn
    992. -
    993. Exercise 3: Normalizing our data
    994. +
    995. Numpy and arrays
    996. +
    997. Matrices in Python
    998. +
    999. Meet the Pandas
    1000. +
    1001.    Simple linear regression model using scikit-learn
    1002. +
    1003.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1004. +
    1005.    Organizing our data
    1006. +
    1007.    And what about using neural networks?
    1008. +
    1009. A first summary
    1010. +
    1011. Why Linear Regression (aka Ordinary Least Squares and family)
    1012. +
    1013. Regression analysis, overarching aims
    1014. +
    1015. Regression analysis, overarching aims II
    1016. +
    1017. Examples
    1018. +
    1019. General linear models
    1020. +
    1021. Rewriting the fitting procedure as a linear algebra problem
    1022. +
    1023. Rewriting the fitting procedure as a linear algebra problem, more details
    1024. +
    1025. Generalizing the fitting procedure as a linear algebra problem
    1026. +
    1027. Generalizing the fitting procedure as a linear algebra problem
    1028. +
    1029. Optimizing our parameters
    1030. +
    1031. Our model for the nuclear binding energies
    1032. +
    1033. Optimizing our parameters, more details
    1034. +
    1035. Interpretations and optimizing our parameters
    1036. +
    1037. Interpretations and optimizing our parameters
    1038. +
    1039. Some useful matrix and vector expressions
    1040. +
    1041. Interpretations and optimizing our parameters
    1042. +
    1043. Own code for Ordinary Least Squares
    1044. +
    1045. Adding error analysis and training set up
    1046. +
    1047. The \( \chi^2 \) function
    1048. +
    1049. The \( \chi^2 \) function
    1050. +
    1051. The \( \chi^2 \) function
    1052. +
    1053. The \( \chi^2 \) function
    1054. +
    1055. The \( \chi^2 \) function
    1056. +
    1057. The \( \chi^2 \) function
    1058. +
    1059. Fitting an Equation of State for Dense Nuclear Matter
    1060. +
    1061. The code
    1062. +
    1063. Splitting our Data in Training and Test data
    1064. +
    1065. Exercises
    1066. +
    1067. Exercise 1: Setting up various Python environments
    1068. +
    1069. Exercise 2: making your own data and exploring scikit-learn
    1070. +
    1071. Exercise 3: Split data in test and training data
    1072. @@ -398,7 +382,7 @@ MathJax.Hub.Config({
    1073. 16
    1074. 17
    1075. ...
    1076. -
    1077. 66
    1078. +
    1079. 62
    1080. »
    1081. diff --git a/doc/pub/week34/html/._week34-bs008.html b/doc/pub/week34/html/._week34-bs008.html index 226078733..837446f48 100644 --- a/doc/pub/week34/html/._week34-bs008.html +++ b/doc/pub/week34/html/._week34-bs008.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    1082. Installing R, C++, cython or Julia
    1083. Installing R, C++, cython, Numba etc
    1084. Numpy examples and Important Matrix and vector handling packages
    1085. -
    1086. Basic Matrix Features
    1087. -
    1088.    Some famous Matrices
    1089. -
    1090.    More Basic Matrix Features
    1091. -
    1092. Numpy and arrays
    1093. -
    1094. Matrices in Python
    1095. -
    1096. Meet the Pandas
    1097. -
    1098. Friday August 27
    1099. -
    1100.    Simple linear regression model using scikit-learn
    1101. -
    1102.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1103. -
    1104.    Organizing our data
    1105. -
    1106.    Seeing the wood for the trees
    1107. -
    1108.    And what about using neural networks?
    1109. -
    1110. A first summary
    1111. -
    1112. Why Linear Regression (aka Ordinary Least Squares and family)
    1113. -
    1114. Regression analysis, overarching aims
    1115. -
    1116. Regression analysis, overarching aims II
    1117. -
    1118. Examples
    1119. -
    1120. General linear models
    1121. -
    1122. Rewriting the fitting procedure as a linear algebra problem
    1123. -
    1124. Rewriting the fitting procedure as a linear algebra problem, more details
    1125. -
    1126. Generalizing the fitting procedure as a linear algebra problem
    1127. -
    1128. Generalizing the fitting procedure as a linear algebra problem
    1129. -
    1130. Optimizing our parameters
    1131. -
    1132. Our model for the nuclear binding energies
    1133. -
    1134. Optimizing our parameters, more details
    1135. -
    1136. Interpretations and optimizing our parameters
    1137. -
    1138. Interpretations and optimizing our parameters
    1139. -
    1140. Some useful matrix and vector expressions
    1141. -
    1142. Interpretations and optimizing our parameters
    1143. -
    1144. Own code for Ordinary Least Squares
    1145. -
    1146. Adding error analysis and training set up
    1147. -
    1148. The \( \chi^2 \) function
    1149. -
    1150. The \( \chi^2 \) function
    1151. -
    1152. The \( \chi^2 \) function
    1153. -
    1154. The \( \chi^2 \) function
    1155. -
    1156. The \( \chi^2 \) function
    1157. -
    1158. The \( \chi^2 \) function
    1159. -
    1160. Fitting an Equation of State for Dense Nuclear Matter
    1161. -
    1162. The code
    1163. -
    1164. Splitting our Data in Training and Test data
    1165. -
    1166. Exercises
    1167. -
    1168. Exercise 1: Setting up various Python environments
    1169. -
    1170. Exercise 2: making your own data and exploring scikit-learn
    1171. -
    1172. Exercise 3: Normalizing our data
    1173. +
    1174. Numpy and arrays
    1175. +
    1176. Matrices in Python
    1177. +
    1178. Meet the Pandas
    1179. +
    1180.    Simple linear regression model using scikit-learn
    1181. +
    1182.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1183. +
    1184.    Organizing our data
    1185. +
    1186.    And what about using neural networks?
    1187. +
    1188. A first summary
    1189. +
    1190. Why Linear Regression (aka Ordinary Least Squares and family)
    1191. +
    1192. Regression analysis, overarching aims
    1193. +
    1194. Regression analysis, overarching aims II
    1195. +
    1196. Examples
    1197. +
    1198. General linear models
    1199. +
    1200. Rewriting the fitting procedure as a linear algebra problem
    1201. +
    1202. Rewriting the fitting procedure as a linear algebra problem, more details
    1203. +
    1204. Generalizing the fitting procedure as a linear algebra problem
    1205. +
    1206. Generalizing the fitting procedure as a linear algebra problem
    1207. +
    1208. Optimizing our parameters
    1209. +
    1210. Our model for the nuclear binding energies
    1211. +
    1212. Optimizing our parameters, more details
    1213. +
    1214. Interpretations and optimizing our parameters
    1215. +
    1216. Interpretations and optimizing our parameters
    1217. +
    1218. Some useful matrix and vector expressions
    1219. +
    1220. Interpretations and optimizing our parameters
    1221. +
    1222. Own code for Ordinary Least Squares
    1223. +
    1224. Adding error analysis and training set up
    1225. +
    1226. The \( \chi^2 \) function
    1227. +
    1228. The \( \chi^2 \) function
    1229. +
    1230. The \( \chi^2 \) function
    1231. +
    1232. The \( \chi^2 \) function
    1233. +
    1234. The \( \chi^2 \) function
    1235. +
    1236. The \( \chi^2 \) function
    1237. +
    1238. Fitting an Equation of State for Dense Nuclear Matter
    1239. +
    1240. The code
    1241. +
    1242. Splitting our Data in Training and Test data
    1243. +
    1244. Exercises
    1245. +
    1246. Exercise 1: Setting up various Python environments
    1247. +
    1248. Exercise 2: making your own data and exploring scikit-learn
    1249. +
    1250. Exercise 3: Split data in test and training data
    1251. @@ -390,7 +374,7 @@ MathJax.Hub.Config({
    1252. 17
    1253. 18
    1254. ...
    1255. -
    1256. 66
    1257. +
    1258. 62
    1259. »
    1260. diff --git a/doc/pub/week34/html/._week34-bs009.html b/doc/pub/week34/html/._week34-bs009.html index 43f68979b..44a93e1f9 100644 --- a/doc/pub/week34/html/._week34-bs009.html +++ b/doc/pub/week34/html/._week34-bs009.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    1261. Installing R, C++, cython or Julia
    1262. Installing R, C++, cython, Numba etc
    1263. Numpy examples and Important Matrix and vector handling packages
    1264. -
    1265. Basic Matrix Features
    1266. -
    1267.    Some famous Matrices
    1268. -
    1269.    More Basic Matrix Features
    1270. -
    1271. Numpy and arrays
    1272. -
    1273. Matrices in Python
    1274. -
    1275. Meet the Pandas
    1276. -
    1277. Friday August 27
    1278. -
    1279.    Simple linear regression model using scikit-learn
    1280. -
    1281.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1282. -
    1283.    Organizing our data
    1284. -
    1285.    Seeing the wood for the trees
    1286. -
    1287.    And what about using neural networks?
    1288. -
    1289. A first summary
    1290. -
    1291. Why Linear Regression (aka Ordinary Least Squares and family)
    1292. -
    1293. Regression analysis, overarching aims
    1294. -
    1295. Regression analysis, overarching aims II
    1296. -
    1297. Examples
    1298. -
    1299. General linear models
    1300. -
    1301. Rewriting the fitting procedure as a linear algebra problem
    1302. -
    1303. Rewriting the fitting procedure as a linear algebra problem, more details
    1304. -
    1305. Generalizing the fitting procedure as a linear algebra problem
    1306. -
    1307. Generalizing the fitting procedure as a linear algebra problem
    1308. -
    1309. Optimizing our parameters
    1310. -
    1311. Our model for the nuclear binding energies
    1312. -
    1313. Optimizing our parameters, more details
    1314. -
    1315. Interpretations and optimizing our parameters
    1316. -
    1317. Interpretations and optimizing our parameters
    1318. -
    1319. Some useful matrix and vector expressions
    1320. -
    1321. Interpretations and optimizing our parameters
    1322. -
    1323. Own code for Ordinary Least Squares
    1324. -
    1325. Adding error analysis and training set up
    1326. -
    1327. The \( \chi^2 \) function
    1328. -
    1329. The \( \chi^2 \) function
    1330. -
    1331. The \( \chi^2 \) function
    1332. -
    1333. The \( \chi^2 \) function
    1334. -
    1335. The \( \chi^2 \) function
    1336. -
    1337. The \( \chi^2 \) function
    1338. -
    1339. Fitting an Equation of State for Dense Nuclear Matter
    1340. -
    1341. The code
    1342. -
    1343. Splitting our Data in Training and Test data
    1344. -
    1345. Exercises
    1346. -
    1347. Exercise 1: Setting up various Python environments
    1348. -
    1349. Exercise 2: making your own data and exploring scikit-learn
    1350. -
    1351. Exercise 3: Normalizing our data
    1352. +
    1353. Numpy and arrays
    1354. +
    1355. Matrices in Python
    1356. +
    1357. Meet the Pandas
    1358. +
    1359.    Simple linear regression model using scikit-learn
    1360. +
    1361.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1362. +
    1363.    Organizing our data
    1364. +
    1365.    And what about using neural networks?
    1366. +
    1367. A first summary
    1368. +
    1369. Why Linear Regression (aka Ordinary Least Squares and family)
    1370. +
    1371. Regression analysis, overarching aims
    1372. +
    1373. Regression analysis, overarching aims II
    1374. +
    1375. Examples
    1376. +
    1377. General linear models
    1378. +
    1379. Rewriting the fitting procedure as a linear algebra problem
    1380. +
    1381. Rewriting the fitting procedure as a linear algebra problem, more details
    1382. +
    1383. Generalizing the fitting procedure as a linear algebra problem
    1384. +
    1385. Generalizing the fitting procedure as a linear algebra problem
    1386. +
    1387. Optimizing our parameters
    1388. +
    1389. Our model for the nuclear binding energies
    1390. +
    1391. Optimizing our parameters, more details
    1392. +
    1393. Interpretations and optimizing our parameters
    1394. +
    1395. Interpretations and optimizing our parameters
    1396. +
    1397. Some useful matrix and vector expressions
    1398. +
    1399. Interpretations and optimizing our parameters
    1400. +
    1401. Own code for Ordinary Least Squares
    1402. +
    1403. Adding error analysis and training set up
    1404. +
    1405. The \( \chi^2 \) function
    1406. +
    1407. The \( \chi^2 \) function
    1408. +
    1409. The \( \chi^2 \) function
    1410. +
    1411. The \( \chi^2 \) function
    1412. +
    1413. The \( \chi^2 \) function
    1414. +
    1415. The \( \chi^2 \) function
    1416. +
    1417. Fitting an Equation of State for Dense Nuclear Matter
    1418. +
    1419. The code
    1420. +
    1421. Splitting our Data in Training and Test data
    1422. +
    1423. Exercises
    1424. +
    1425. Exercise 1: Setting up various Python environments
    1426. +
    1427. Exercise 2: making your own data and exploring scikit-learn
    1428. +
    1429. Exercise 3: Split data in test and training data
    1430. @@ -392,7 +376,7 @@ MathJax.Hub.Config({
    1431. 18
    1432. 19
    1433. ...
    1434. -
    1435. 66
    1436. +
    1437. 62
    1438. »
    1439. diff --git a/doc/pub/week34/html/._week34-bs010.html b/doc/pub/week34/html/._week34-bs010.html index dd8d98177..f4eb43050 100644 --- a/doc/pub/week34/html/._week34-bs010.html +++ b/doc/pub/week34/html/._week34-bs010.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    1440. Installing R, C++, cython or Julia
    1441. Installing R, C++, cython, Numba etc
    1442. Numpy examples and Important Matrix and vector handling packages
    1443. -
    1444. Basic Matrix Features
    1445. -
    1446.    Some famous Matrices
    1447. -
    1448.    More Basic Matrix Features
    1449. -
    1450. Numpy and arrays
    1451. -
    1452. Matrices in Python
    1453. -
    1454. Meet the Pandas
    1455. -
    1456. Friday August 27
    1457. -
    1458.    Simple linear regression model using scikit-learn
    1459. -
    1460.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1461. -
    1462.    Organizing our data
    1463. -
    1464.    Seeing the wood for the trees
    1465. -
    1466.    And what about using neural networks?
    1467. -
    1468. A first summary
    1469. -
    1470. Why Linear Regression (aka Ordinary Least Squares and family)
    1471. -
    1472. Regression analysis, overarching aims
    1473. -
    1474. Regression analysis, overarching aims II
    1475. -
    1476. Examples
    1477. -
    1478. General linear models
    1479. -
    1480. Rewriting the fitting procedure as a linear algebra problem
    1481. -
    1482. Rewriting the fitting procedure as a linear algebra problem, more details
    1483. -
    1484. Generalizing the fitting procedure as a linear algebra problem
    1485. -
    1486. Generalizing the fitting procedure as a linear algebra problem
    1487. -
    1488. Optimizing our parameters
    1489. -
    1490. Our model for the nuclear binding energies
    1491. -
    1492. Optimizing our parameters, more details
    1493. -
    1494. Interpretations and optimizing our parameters
    1495. -
    1496. Interpretations and optimizing our parameters
    1497. -
    1498. Some useful matrix and vector expressions
    1499. -
    1500. Interpretations and optimizing our parameters
    1501. -
    1502. Own code for Ordinary Least Squares
    1503. -
    1504. Adding error analysis and training set up
    1505. -
    1506. The \( \chi^2 \) function
    1507. -
    1508. The \( \chi^2 \) function
    1509. -
    1510. The \( \chi^2 \) function
    1511. -
    1512. The \( \chi^2 \) function
    1513. -
    1514. The \( \chi^2 \) function
    1515. -
    1516. The \( \chi^2 \) function
    1517. -
    1518. Fitting an Equation of State for Dense Nuclear Matter
    1519. -
    1520. The code
    1521. -
    1522. Splitting our Data in Training and Test data
    1523. -
    1524. Exercises
    1525. -
    1526. Exercise 1: Setting up various Python environments
    1527. -
    1528. Exercise 2: making your own data and exploring scikit-learn
    1529. -
    1530. Exercise 3: Normalizing our data
    1531. +
    1532. Numpy and arrays
    1533. +
    1534. Matrices in Python
    1535. +
    1536. Meet the Pandas
    1537. +
    1538.    Simple linear regression model using scikit-learn
    1539. +
    1540.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1541. +
    1542.    Organizing our data
    1543. +
    1544.    And what about using neural networks?
    1545. +
    1546. A first summary
    1547. +
    1548. Why Linear Regression (aka Ordinary Least Squares and family)
    1549. +
    1550. Regression analysis, overarching aims
    1551. +
    1552. Regression analysis, overarching aims II
    1553. +
    1554. Examples
    1555. +
    1556. General linear models
    1557. +
    1558. Rewriting the fitting procedure as a linear algebra problem
    1559. +
    1560. Rewriting the fitting procedure as a linear algebra problem, more details
    1561. +
    1562. Generalizing the fitting procedure as a linear algebra problem
    1563. +
    1564. Generalizing the fitting procedure as a linear algebra problem
    1565. +
    1566. Optimizing our parameters
    1567. +
    1568. Our model for the nuclear binding energies
    1569. +
    1570. Optimizing our parameters, more details
    1571. +
    1572. Interpretations and optimizing our parameters
    1573. +
    1574. Interpretations and optimizing our parameters
    1575. +
    1576. Some useful matrix and vector expressions
    1577. +
    1578. Interpretations and optimizing our parameters
    1579. +
    1580. Own code for Ordinary Least Squares
    1581. +
    1582. Adding error analysis and training set up
    1583. +
    1584. The \( \chi^2 \) function
    1585. +
    1586. The \( \chi^2 \) function
    1587. +
    1588. The \( \chi^2 \) function
    1589. +
    1590. The \( \chi^2 \) function
    1591. +
    1592. The \( \chi^2 \) function
    1593. +
    1594. The \( \chi^2 \) function
    1595. +
    1596. Fitting an Equation of State for Dense Nuclear Matter
    1597. +
    1598. The code
    1599. +
    1600. Splitting our Data in Training and Test data
    1601. +
    1602. Exercises
    1603. +
    1604. Exercise 1: Setting up various Python environments
    1605. +
    1606. Exercise 2: making your own data and exploring scikit-learn
    1607. +
    1608. Exercise 3: Split data in test and training data
    1609. @@ -389,7 +373,7 @@ Python is the recurring programming language.
    1610. 19
    1611. 20
    1612. ...
    1613. -
    1614. 66
    1615. +
    1616. 62
    1617. »
    1618. diff --git a/doc/pub/week34/html/._week34-bs011.html b/doc/pub/week34/html/._week34-bs011.html index 7a0f68b07..bb0e8db2e 100644 --- a/doc/pub/week34/html/._week34-bs011.html +++ b/doc/pub/week34/html/._week34-bs011.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    1619. Installing R, C++, cython or Julia
    1620. Installing R, C++, cython, Numba etc
    1621. Numpy examples and Important Matrix and vector handling packages
    1622. -
    1623. Basic Matrix Features
    1624. -
    1625.    Some famous Matrices
    1626. -
    1627.    More Basic Matrix Features
    1628. -
    1629. Numpy and arrays
    1630. -
    1631. Matrices in Python
    1632. -
    1633. Meet the Pandas
    1634. -
    1635. Friday August 27
    1636. -
    1637.    Simple linear regression model using scikit-learn
    1638. -
    1639.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1640. -
    1641.    Organizing our data
    1642. -
    1643.    Seeing the wood for the trees
    1644. -
    1645.    And what about using neural networks?
    1646. -
    1647. A first summary
    1648. -
    1649. Why Linear Regression (aka Ordinary Least Squares and family)
    1650. -
    1651. Regression analysis, overarching aims
    1652. -
    1653. Regression analysis, overarching aims II
    1654. -
    1655. Examples
    1656. -
    1657. General linear models
    1658. -
    1659. Rewriting the fitting procedure as a linear algebra problem
    1660. -
    1661. Rewriting the fitting procedure as a linear algebra problem, more details
    1662. -
    1663. Generalizing the fitting procedure as a linear algebra problem
    1664. -
    1665. Generalizing the fitting procedure as a linear algebra problem
    1666. -
    1667. Optimizing our parameters
    1668. -
    1669. Our model for the nuclear binding energies
    1670. -
    1671. Optimizing our parameters, more details
    1672. -
    1673. Interpretations and optimizing our parameters
    1674. -
    1675. Interpretations and optimizing our parameters
    1676. -
    1677. Some useful matrix and vector expressions
    1678. -
    1679. Interpretations and optimizing our parameters
    1680. -
    1681. Own code for Ordinary Least Squares
    1682. -
    1683. Adding error analysis and training set up
    1684. -
    1685. The \( \chi^2 \) function
    1686. -
    1687. The \( \chi^2 \) function
    1688. -
    1689. The \( \chi^2 \) function
    1690. -
    1691. The \( \chi^2 \) function
    1692. -
    1693. The \( \chi^2 \) function
    1694. -
    1695. The \( \chi^2 \) function
    1696. -
    1697. Fitting an Equation of State for Dense Nuclear Matter
    1698. -
    1699. The code
    1700. -
    1701. Splitting our Data in Training and Test data
    1702. -
    1703. Exercises
    1704. -
    1705. Exercise 1: Setting up various Python environments
    1706. -
    1707. Exercise 2: making your own data and exploring scikit-learn
    1708. -
    1709. Exercise 3: Normalizing our data
    1710. +
    1711. Numpy and arrays
    1712. +
    1713. Matrices in Python
    1714. +
    1715. Meet the Pandas
    1716. +
    1717.    Simple linear regression model using scikit-learn
    1718. +
    1719.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1720. +
    1721.    Organizing our data
    1722. +
    1723.    And what about using neural networks?
    1724. +
    1725. A first summary
    1726. +
    1727. Why Linear Regression (aka Ordinary Least Squares and family)
    1728. +
    1729. Regression analysis, overarching aims
    1730. +
    1731. Regression analysis, overarching aims II
    1732. +
    1733. Examples
    1734. +
    1735. General linear models
    1736. +
    1737. Rewriting the fitting procedure as a linear algebra problem
    1738. +
    1739. Rewriting the fitting procedure as a linear algebra problem, more details
    1740. +
    1741. Generalizing the fitting procedure as a linear algebra problem
    1742. +
    1743. Generalizing the fitting procedure as a linear algebra problem
    1744. +
    1745. Optimizing our parameters
    1746. +
    1747. Our model for the nuclear binding energies
    1748. +
    1749. Optimizing our parameters, more details
    1750. +
    1751. Interpretations and optimizing our parameters
    1752. +
    1753. Interpretations and optimizing our parameters
    1754. +
    1755. Some useful matrix and vector expressions
    1756. +
    1757. Interpretations and optimizing our parameters
    1758. +
    1759. Own code for Ordinary Least Squares
    1760. +
    1761. Adding error analysis and training set up
    1762. +
    1763. The \( \chi^2 \) function
    1764. +
    1765. The \( \chi^2 \) function
    1766. +
    1767. The \( \chi^2 \) function
    1768. +
    1769. The \( \chi^2 \) function
    1770. +
    1771. The \( \chi^2 \) function
    1772. +
    1773. The \( \chi^2 \) function
    1774. +
    1775. Fitting an Equation of State for Dense Nuclear Matter
    1776. +
    1777. The code
    1778. +
    1779. Splitting our Data in Training and Test data
    1780. +
    1781. Exercises
    1782. +
    1783. Exercise 1: Setting up various Python environments
    1784. +
    1785. Exercise 2: making your own data and exploring scikit-learn
    1786. +
    1787. Exercise 3: Split data in test and training data
    1788. @@ -413,7 +397,7 @@ specifically, after this course you will
    1789. 20
    1790. 21
    1791. ...
    1792. -
    1793. 66
    1794. +
    1795. 62
    1796. »
    1797. diff --git a/doc/pub/week34/html/._week34-bs012.html b/doc/pub/week34/html/._week34-bs012.html index 63566c2fc..e1297ade0 100644 --- a/doc/pub/week34/html/._week34-bs012.html +++ b/doc/pub/week34/html/._week34-bs012.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    1798. Installing R, C++, cython or Julia
    1799. Installing R, C++, cython, Numba etc
    1800. Numpy examples and Important Matrix and vector handling packages
    1801. -
    1802. Basic Matrix Features
    1803. -
    1804.    Some famous Matrices
    1805. -
    1806.    More Basic Matrix Features
    1807. -
    1808. Numpy and arrays
    1809. -
    1810. Matrices in Python
    1811. -
    1812. Meet the Pandas
    1813. -
    1814. Friday August 27
    1815. -
    1816.    Simple linear regression model using scikit-learn
    1817. -
    1818.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1819. -
    1820.    Organizing our data
    1821. -
    1822.    Seeing the wood for the trees
    1823. -
    1824.    And what about using neural networks?
    1825. -
    1826. A first summary
    1827. -
    1828. Why Linear Regression (aka Ordinary Least Squares and family)
    1829. -
    1830. Regression analysis, overarching aims
    1831. -
    1832. Regression analysis, overarching aims II
    1833. -
    1834. Examples
    1835. -
    1836. General linear models
    1837. -
    1838. Rewriting the fitting procedure as a linear algebra problem
    1839. -
    1840. Rewriting the fitting procedure as a linear algebra problem, more details
    1841. -
    1842. Generalizing the fitting procedure as a linear algebra problem
    1843. -
    1844. Generalizing the fitting procedure as a linear algebra problem
    1845. -
    1846. Optimizing our parameters
    1847. -
    1848. Our model for the nuclear binding energies
    1849. -
    1850. Optimizing our parameters, more details
    1851. -
    1852. Interpretations and optimizing our parameters
    1853. -
    1854. Interpretations and optimizing our parameters
    1855. -
    1856. Some useful matrix and vector expressions
    1857. -
    1858. Interpretations and optimizing our parameters
    1859. -
    1860. Own code for Ordinary Least Squares
    1861. -
    1862. Adding error analysis and training set up
    1863. -
    1864. The \( \chi^2 \) function
    1865. -
    1866. The \( \chi^2 \) function
    1867. -
    1868. The \( \chi^2 \) function
    1869. -
    1870. The \( \chi^2 \) function
    1871. -
    1872. The \( \chi^2 \) function
    1873. -
    1874. The \( \chi^2 \) function
    1875. -
    1876. Fitting an Equation of State for Dense Nuclear Matter
    1877. -
    1878. The code
    1879. -
    1880. Splitting our Data in Training and Test data
    1881. -
    1882. Exercises
    1883. -
    1884. Exercise 1: Setting up various Python environments
    1885. -
    1886. Exercise 2: making your own data and exploring scikit-learn
    1887. -
    1888. Exercise 3: Normalizing our data
    1889. +
    1890. Numpy and arrays
    1891. +
    1892. Matrices in Python
    1893. +
    1894. Meet the Pandas
    1895. +
    1896.    Simple linear regression model using scikit-learn
    1897. +
    1898.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1899. +
    1900.    Organizing our data
    1901. +
    1902.    And what about using neural networks?
    1903. +
    1904. A first summary
    1905. +
    1906. Why Linear Regression (aka Ordinary Least Squares and family)
    1907. +
    1908. Regression analysis, overarching aims
    1909. +
    1910. Regression analysis, overarching aims II
    1911. +
    1912. Examples
    1913. +
    1914. General linear models
    1915. +
    1916. Rewriting the fitting procedure as a linear algebra problem
    1917. +
    1918. Rewriting the fitting procedure as a linear algebra problem, more details
    1919. +
    1920. Generalizing the fitting procedure as a linear algebra problem
    1921. +
    1922. Generalizing the fitting procedure as a linear algebra problem
    1923. +
    1924. Optimizing our parameters
    1925. +
    1926. Our model for the nuclear binding energies
    1927. +
    1928. Optimizing our parameters, more details
    1929. +
    1930. Interpretations and optimizing our parameters
    1931. +
    1932. Interpretations and optimizing our parameters
    1933. +
    1934. Some useful matrix and vector expressions
    1935. +
    1936. Interpretations and optimizing our parameters
    1937. +
    1938. Own code for Ordinary Least Squares
    1939. +
    1940. Adding error analysis and training set up
    1941. +
    1942. The \( \chi^2 \) function
    1943. +
    1944. The \( \chi^2 \) function
    1945. +
    1946. The \( \chi^2 \) function
    1947. +
    1948. The \( \chi^2 \) function
    1949. +
    1950. The \( \chi^2 \) function
    1951. +
    1952. The \( \chi^2 \) function
    1953. +
    1954. Fitting an Equation of State for Dense Nuclear Matter
    1955. +
    1956. The code
    1957. +
    1958. Splitting our Data in Training and Test data
    1959. +
    1960. Exercises
    1961. +
    1962. Exercise 1: Setting up various Python environments
    1963. +
    1964. Exercise 2: making your own data and exploring scikit-learn
    1965. +
    1966. Exercise 3: Split data in test and training data
    1967. @@ -404,7 +388,7 @@ MathJax.Hub.Config({
    1968. 21
    1969. 22
    1970. ...
    1971. -
    1972. 66
    1973. +
    1974. 62
    1975. »
    1976. diff --git a/doc/pub/week34/html/._week34-bs013.html b/doc/pub/week34/html/._week34-bs013.html index 8049abdfa..139c98ceb 100644 --- a/doc/pub/week34/html/._week34-bs013.html +++ b/doc/pub/week34/html/._week34-bs013.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    1977. Installing R, C++, cython or Julia
    1978. Installing R, C++, cython, Numba etc
    1979. Numpy examples and Important Matrix and vector handling packages
    1980. -
    1981. Basic Matrix Features
    1982. -
    1983.    Some famous Matrices
    1984. -
    1985.    More Basic Matrix Features
    1986. -
    1987. Numpy and arrays
    1988. -
    1989. Matrices in Python
    1990. -
    1991. Meet the Pandas
    1992. -
    1993. Friday August 27
    1994. -
    1995.    Simple linear regression model using scikit-learn
    1996. -
    1997.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    1998. -
    1999.    Organizing our data
    2000. -
    2001.    Seeing the wood for the trees
    2002. -
    2003.    And what about using neural networks?
    2004. -
    2005. A first summary
    2006. -
    2007. Why Linear Regression (aka Ordinary Least Squares and family)
    2008. -
    2009. Regression analysis, overarching aims
    2010. -
    2011. Regression analysis, overarching aims II
    2012. -
    2013. Examples
    2014. -
    2015. General linear models
    2016. -
    2017. Rewriting the fitting procedure as a linear algebra problem
    2018. -
    2019. Rewriting the fitting procedure as a linear algebra problem, more details
    2020. -
    2021. Generalizing the fitting procedure as a linear algebra problem
    2022. -
    2023. Generalizing the fitting procedure as a linear algebra problem
    2024. -
    2025. Optimizing our parameters
    2026. -
    2027. Our model for the nuclear binding energies
    2028. -
    2029. Optimizing our parameters, more details
    2030. -
    2031. Interpretations and optimizing our parameters
    2032. -
    2033. Interpretations and optimizing our parameters
    2034. -
    2035. Some useful matrix and vector expressions
    2036. -
    2037. Interpretations and optimizing our parameters
    2038. -
    2039. Own code for Ordinary Least Squares
    2040. -
    2041. Adding error analysis and training set up
    2042. -
    2043. The \( \chi^2 \) function
    2044. -
    2045. The \( \chi^2 \) function
    2046. -
    2047. The \( \chi^2 \) function
    2048. -
    2049. The \( \chi^2 \) function
    2050. -
    2051. The \( \chi^2 \) function
    2052. -
    2053. The \( \chi^2 \) function
    2054. -
    2055. Fitting an Equation of State for Dense Nuclear Matter
    2056. -
    2057. The code
    2058. -
    2059. Splitting our Data in Training and Test data
    2060. -
    2061. Exercises
    2062. -
    2063. Exercise 1: Setting up various Python environments
    2064. -
    2065. Exercise 2: making your own data and exploring scikit-learn
    2066. -
    2067. Exercise 3: Normalizing our data
    2068. +
    2069. Numpy and arrays
    2070. +
    2071. Matrices in Python
    2072. +
    2073. Meet the Pandas
    2074. +
    2075.    Simple linear regression model using scikit-learn
    2076. +
    2077.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2078. +
    2079.    Organizing our data
    2080. +
    2081.    And what about using neural networks?
    2082. +
    2083. A first summary
    2084. +
    2085. Why Linear Regression (aka Ordinary Least Squares and family)
    2086. +
    2087. Regression analysis, overarching aims
    2088. +
    2089. Regression analysis, overarching aims II
    2090. +
    2091. Examples
    2092. +
    2093. General linear models
    2094. +
    2095. Rewriting the fitting procedure as a linear algebra problem
    2096. +
    2097. Rewriting the fitting procedure as a linear algebra problem, more details
    2098. +
    2099. Generalizing the fitting procedure as a linear algebra problem
    2100. +
    2101. Generalizing the fitting procedure as a linear algebra problem
    2102. +
    2103. Optimizing our parameters
    2104. +
    2105. Our model for the nuclear binding energies
    2106. +
    2107. Optimizing our parameters, more details
    2108. +
    2109. Interpretations and optimizing our parameters
    2110. +
    2111. Interpretations and optimizing our parameters
    2112. +
    2113. Some useful matrix and vector expressions
    2114. +
    2115. Interpretations and optimizing our parameters
    2116. +
    2117. Own code for Ordinary Least Squares
    2118. +
    2119. Adding error analysis and training set up
    2120. +
    2121. The \( \chi^2 \) function
    2122. +
    2123. The \( \chi^2 \) function
    2124. +
    2125. The \( \chi^2 \) function
    2126. +
    2127. The \( \chi^2 \) function
    2128. +
    2129. The \( \chi^2 \) function
    2130. +
    2131. The \( \chi^2 \) function
    2132. +
    2133. Fitting an Equation of State for Dense Nuclear Matter
    2134. +
    2135. The code
    2136. +
    2137. Splitting our Data in Training and Test data
    2138. +
    2139. Exercises
    2140. +
    2141. Exercise 1: Setting up various Python environments
    2142. +
    2143. Exercise 2: making your own data and exploring scikit-learn
    2144. +
    2145. Exercise 3: Split data in test and training data
    2146. @@ -406,7 +390,7 @@ MathJax.Hub.Config({
    2147. 22
    2148. 23
    2149. ...
    2150. -
    2151. 66
    2152. +
    2153. 62
    2154. »
    2155. diff --git a/doc/pub/week34/html/._week34-bs014.html b/doc/pub/week34/html/._week34-bs014.html index 1de18d9e5..8fe5ee44c 100644 --- a/doc/pub/week34/html/._week34-bs014.html +++ b/doc/pub/week34/html/._week34-bs014.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    2156. Installing R, C++, cython or Julia
    2157. Installing R, C++, cython, Numba etc
    2158. Numpy examples and Important Matrix and vector handling packages
    2159. -
    2160. Basic Matrix Features
    2161. -
    2162.    Some famous Matrices
    2163. -
    2164.    More Basic Matrix Features
    2165. -
    2166. Numpy and arrays
    2167. -
    2168. Matrices in Python
    2169. -
    2170. Meet the Pandas
    2171. -
    2172. Friday August 27
    2173. -
    2174.    Simple linear regression model using scikit-learn
    2175. -
    2176.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2177. -
    2178.    Organizing our data
    2179. -
    2180.    Seeing the wood for the trees
    2181. -
    2182.    And what about using neural networks?
    2183. -
    2184. A first summary
    2185. -
    2186. Why Linear Regression (aka Ordinary Least Squares and family)
    2187. -
    2188. Regression analysis, overarching aims
    2189. -
    2190. Regression analysis, overarching aims II
    2191. -
    2192. Examples
    2193. -
    2194. General linear models
    2195. -
    2196. Rewriting the fitting procedure as a linear algebra problem
    2197. -
    2198. Rewriting the fitting procedure as a linear algebra problem, more details
    2199. -
    2200. Generalizing the fitting procedure as a linear algebra problem
    2201. -
    2202. Generalizing the fitting procedure as a linear algebra problem
    2203. -
    2204. Optimizing our parameters
    2205. -
    2206. Our model for the nuclear binding energies
    2207. -
    2208. Optimizing our parameters, more details
    2209. -
    2210. Interpretations and optimizing our parameters
    2211. -
    2212. Interpretations and optimizing our parameters
    2213. -
    2214. Some useful matrix and vector expressions
    2215. -
    2216. Interpretations and optimizing our parameters
    2217. -
    2218. Own code for Ordinary Least Squares
    2219. -
    2220. Adding error analysis and training set up
    2221. -
    2222. The \( \chi^2 \) function
    2223. -
    2224. The \( \chi^2 \) function
    2225. -
    2226. The \( \chi^2 \) function
    2227. -
    2228. The \( \chi^2 \) function
    2229. -
    2230. The \( \chi^2 \) function
    2231. -
    2232. The \( \chi^2 \) function
    2233. -
    2234. Fitting an Equation of State for Dense Nuclear Matter
    2235. -
    2236. The code
    2237. -
    2238. Splitting our Data in Training and Test data
    2239. -
    2240. Exercises
    2241. -
    2242. Exercise 1: Setting up various Python environments
    2243. -
    2244. Exercise 2: making your own data and exploring scikit-learn
    2245. -
    2246. Exercise 3: Normalizing our data
    2247. +
    2248. Numpy and arrays
    2249. +
    2250. Matrices in Python
    2251. +
    2252. Meet the Pandas
    2253. +
    2254.    Simple linear regression model using scikit-learn
    2255. +
    2256.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2257. +
    2258.    Organizing our data
    2259. +
    2260.    And what about using neural networks?
    2261. +
    2262. A first summary
    2263. +
    2264. Why Linear Regression (aka Ordinary Least Squares and family)
    2265. +
    2266. Regression analysis, overarching aims
    2267. +
    2268. Regression analysis, overarching aims II
    2269. +
    2270. Examples
    2271. +
    2272. General linear models
    2273. +
    2274. Rewriting the fitting procedure as a linear algebra problem
    2275. +
    2276. Rewriting the fitting procedure as a linear algebra problem, more details
    2277. +
    2278. Generalizing the fitting procedure as a linear algebra problem
    2279. +
    2280. Generalizing the fitting procedure as a linear algebra problem
    2281. +
    2282. Optimizing our parameters
    2283. +
    2284. Our model for the nuclear binding energies
    2285. +
    2286. Optimizing our parameters, more details
    2287. +
    2288. Interpretations and optimizing our parameters
    2289. +
    2290. Interpretations and optimizing our parameters
    2291. +
    2292. Some useful matrix and vector expressions
    2293. +
    2294. Interpretations and optimizing our parameters
    2295. +
    2296. Own code for Ordinary Least Squares
    2297. +
    2298. Adding error analysis and training set up
    2299. +
    2300. The \( \chi^2 \) function
    2301. +
    2302. The \( \chi^2 \) function
    2303. +
    2304. The \( \chi^2 \) function
    2305. +
    2306. The \( \chi^2 \) function
    2307. +
    2308. The \( \chi^2 \) function
    2309. +
    2310. The \( \chi^2 \) function
    2311. +
    2312. Fitting an Equation of State for Dense Nuclear Matter
    2313. +
    2314. The code
    2315. +
    2316. Splitting our Data in Training and Test data
    2317. +
    2318. Exercises
    2319. +
    2320. Exercise 1: Setting up various Python environments
    2321. +
    2322. Exercise 2: making your own data and exploring scikit-learn
    2323. +
    2324. Exercise 3: Split data in test and training data
    2325. @@ -389,7 +373,7 @@ MathJax.Hub.Config({
    2326. 23
    2327. 24
    2328. ...
    2329. -
    2330. 66
    2331. +
    2332. 62
    2333. »
    2334. diff --git a/doc/pub/week34/html/._week34-bs015.html b/doc/pub/week34/html/._week34-bs015.html index b67e2f56e..9c9a00303 100644 --- a/doc/pub/week34/html/._week34-bs015.html +++ b/doc/pub/week34/html/._week34-bs015.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    2335. Installing R, C++, cython or Julia
    2336. Installing R, C++, cython, Numba etc
    2337. Numpy examples and Important Matrix and vector handling packages
    2338. -
    2339. Basic Matrix Features
    2340. -
    2341.    Some famous Matrices
    2342. -
    2343.    More Basic Matrix Features
    2344. -
    2345. Numpy and arrays
    2346. -
    2347. Matrices in Python
    2348. -
    2349. Meet the Pandas
    2350. -
    2351. Friday August 27
    2352. -
    2353.    Simple linear regression model using scikit-learn
    2354. -
    2355.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2356. -
    2357.    Organizing our data
    2358. -
    2359.    Seeing the wood for the trees
    2360. -
    2361.    And what about using neural networks?
    2362. -
    2363. A first summary
    2364. -
    2365. Why Linear Regression (aka Ordinary Least Squares and family)
    2366. -
    2367. Regression analysis, overarching aims
    2368. -
    2369. Regression analysis, overarching aims II
    2370. -
    2371. Examples
    2372. -
    2373. General linear models
    2374. -
    2375. Rewriting the fitting procedure as a linear algebra problem
    2376. -
    2377. Rewriting the fitting procedure as a linear algebra problem, more details
    2378. -
    2379. Generalizing the fitting procedure as a linear algebra problem
    2380. -
    2381. Generalizing the fitting procedure as a linear algebra problem
    2382. -
    2383. Optimizing our parameters
    2384. -
    2385. Our model for the nuclear binding energies
    2386. -
    2387. Optimizing our parameters, more details
    2388. -
    2389. Interpretations and optimizing our parameters
    2390. -
    2391. Interpretations and optimizing our parameters
    2392. -
    2393. Some useful matrix and vector expressions
    2394. -
    2395. Interpretations and optimizing our parameters
    2396. -
    2397. Own code for Ordinary Least Squares
    2398. -
    2399. Adding error analysis and training set up
    2400. -
    2401. The \( \chi^2 \) function
    2402. -
    2403. The \( \chi^2 \) function
    2404. -
    2405. The \( \chi^2 \) function
    2406. -
    2407. The \( \chi^2 \) function
    2408. -
    2409. The \( \chi^2 \) function
    2410. -
    2411. The \( \chi^2 \) function
    2412. -
    2413. Fitting an Equation of State for Dense Nuclear Matter
    2414. -
    2415. The code
    2416. -
    2417. Splitting our Data in Training and Test data
    2418. -
    2419. Exercises
    2420. -
    2421. Exercise 1: Setting up various Python environments
    2422. -
    2423. Exercise 2: making your own data and exploring scikit-learn
    2424. -
    2425. Exercise 3: Normalizing our data
    2426. +
    2427. Numpy and arrays
    2428. +
    2429. Matrices in Python
    2430. +
    2431. Meet the Pandas
    2432. +
    2433.    Simple linear regression model using scikit-learn
    2434. +
    2435.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2436. +
    2437.    Organizing our data
    2438. +
    2439.    And what about using neural networks?
    2440. +
    2441. A first summary
    2442. +
    2443. Why Linear Regression (aka Ordinary Least Squares and family)
    2444. +
    2445. Regression analysis, overarching aims
    2446. +
    2447. Regression analysis, overarching aims II
    2448. +
    2449. Examples
    2450. +
    2451. General linear models
    2452. +
    2453. Rewriting the fitting procedure as a linear algebra problem
    2454. +
    2455. Rewriting the fitting procedure as a linear algebra problem, more details
    2456. +
    2457. Generalizing the fitting procedure as a linear algebra problem
    2458. +
    2459. Generalizing the fitting procedure as a linear algebra problem
    2460. +
    2461. Optimizing our parameters
    2462. +
    2463. Our model for the nuclear binding energies
    2464. +
    2465. Optimizing our parameters, more details
    2466. +
    2467. Interpretations and optimizing our parameters
    2468. +
    2469. Interpretations and optimizing our parameters
    2470. +
    2471. Some useful matrix and vector expressions
    2472. +
    2473. Interpretations and optimizing our parameters
    2474. +
    2475. Own code for Ordinary Least Squares
    2476. +
    2477. Adding error analysis and training set up
    2478. +
    2479. The \( \chi^2 \) function
    2480. +
    2481. The \( \chi^2 \) function
    2482. +
    2483. The \( \chi^2 \) function
    2484. +
    2485. The \( \chi^2 \) function
    2486. +
    2487. The \( \chi^2 \) function
    2488. +
    2489. The \( \chi^2 \) function
    2490. +
    2491. Fitting an Equation of State for Dense Nuclear Matter
    2492. +
    2493. The code
    2494. +
    2495. Splitting our Data in Training and Test data
    2496. +
    2497. Exercises
    2498. +
    2499. Exercise 1: Setting up various Python environments
    2500. +
    2501. Exercise 2: making your own data and exploring scikit-learn
    2502. +
    2503. Exercise 3: Split data in test and training data
    2504. @@ -391,7 +375,7 @@ MathJax.Hub.Config({
    2505. 24
    2506. 25
    2507. ...
    2508. -
    2509. 66
    2510. +
    2511. 62
    2512. »
    2513. diff --git a/doc/pub/week34/html/._week34-bs016.html b/doc/pub/week34/html/._week34-bs016.html index fb8b304b8..99ee3dbc2 100644 --- a/doc/pub/week34/html/._week34-bs016.html +++ b/doc/pub/week34/html/._week34-bs016.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    2514. Installing R, C++, cython or Julia
    2515. Installing R, C++, cython, Numba etc
    2516. Numpy examples and Important Matrix and vector handling packages
    2517. -
    2518. Basic Matrix Features
    2519. -
    2520.    Some famous Matrices
    2521. -
    2522.    More Basic Matrix Features
    2523. -
    2524. Numpy and arrays
    2525. -
    2526. Matrices in Python
    2527. -
    2528. Meet the Pandas
    2529. -
    2530. Friday August 27
    2531. -
    2532.    Simple linear regression model using scikit-learn
    2533. -
    2534.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2535. -
    2536.    Organizing our data
    2537. -
    2538.    Seeing the wood for the trees
    2539. -
    2540.    And what about using neural networks?
    2541. -
    2542. A first summary
    2543. -
    2544. Why Linear Regression (aka Ordinary Least Squares and family)
    2545. -
    2546. Regression analysis, overarching aims
    2547. -
    2548. Regression analysis, overarching aims II
    2549. -
    2550. Examples
    2551. -
    2552. General linear models
    2553. -
    2554. Rewriting the fitting procedure as a linear algebra problem
    2555. -
    2556. Rewriting the fitting procedure as a linear algebra problem, more details
    2557. -
    2558. Generalizing the fitting procedure as a linear algebra problem
    2559. -
    2560. Generalizing the fitting procedure as a linear algebra problem
    2561. -
    2562. Optimizing our parameters
    2563. -
    2564. Our model for the nuclear binding energies
    2565. -
    2566. Optimizing our parameters, more details
    2567. -
    2568. Interpretations and optimizing our parameters
    2569. -
    2570. Interpretations and optimizing our parameters
    2571. -
    2572. Some useful matrix and vector expressions
    2573. -
    2574. Interpretations and optimizing our parameters
    2575. -
    2576. Own code for Ordinary Least Squares
    2577. -
    2578. Adding error analysis and training set up
    2579. -
    2580. The \( \chi^2 \) function
    2581. -
    2582. The \( \chi^2 \) function
    2583. -
    2584. The \( \chi^2 \) function
    2585. -
    2586. The \( \chi^2 \) function
    2587. -
    2588. The \( \chi^2 \) function
    2589. -
    2590. The \( \chi^2 \) function
    2591. -
    2592. Fitting an Equation of State for Dense Nuclear Matter
    2593. -
    2594. The code
    2595. -
    2596. Splitting our Data in Training and Test data
    2597. -
    2598. Exercises
    2599. -
    2600. Exercise 1: Setting up various Python environments
    2601. -
    2602. Exercise 2: making your own data and exploring scikit-learn
    2603. -
    2604. Exercise 3: Normalizing our data
    2605. +
    2606. Numpy and arrays
    2607. +
    2608. Matrices in Python
    2609. +
    2610. Meet the Pandas
    2611. +
    2612.    Simple linear regression model using scikit-learn
    2613. +
    2614.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2615. +
    2616.    Organizing our data
    2617. +
    2618.    And what about using neural networks?
    2619. +
    2620. A first summary
    2621. +
    2622. Why Linear Regression (aka Ordinary Least Squares and family)
    2623. +
    2624. Regression analysis, overarching aims
    2625. +
    2626. Regression analysis, overarching aims II
    2627. +
    2628. Examples
    2629. +
    2630. General linear models
    2631. +
    2632. Rewriting the fitting procedure as a linear algebra problem
    2633. +
    2634. Rewriting the fitting procedure as a linear algebra problem, more details
    2635. +
    2636. Generalizing the fitting procedure as a linear algebra problem
    2637. +
    2638. Generalizing the fitting procedure as a linear algebra problem
    2639. +
    2640. Optimizing our parameters
    2641. +
    2642. Our model for the nuclear binding energies
    2643. +
    2644. Optimizing our parameters, more details
    2645. +
    2646. Interpretations and optimizing our parameters
    2647. +
    2648. Interpretations and optimizing our parameters
    2649. +
    2650. Some useful matrix and vector expressions
    2651. +
    2652. Interpretations and optimizing our parameters
    2653. +
    2654. Own code for Ordinary Least Squares
    2655. +
    2656. Adding error analysis and training set up
    2657. +
    2658. The \( \chi^2 \) function
    2659. +
    2660. The \( \chi^2 \) function
    2661. +
    2662. The \( \chi^2 \) function
    2663. +
    2664. The \( \chi^2 \) function
    2665. +
    2666. The \( \chi^2 \) function
    2667. +
    2668. The \( \chi^2 \) function
    2669. +
    2670. Fitting an Equation of State for Dense Nuclear Matter
    2671. +
    2672. The code
    2673. +
    2674. Splitting our Data in Training and Test data
    2675. +
    2676. Exercises
    2677. +
    2678. Exercise 1: Setting up various Python environments
    2679. +
    2680. Exercise 2: making your own data and exploring scikit-learn
    2681. +
    2682. Exercise 3: Split data in test and training data
    2683. @@ -382,9 +366,9 @@ topics and tools as well as showing the power of various Python libraries for machine learning and statistical data analysis.

      -

      Here, we will mainly focus on two +

      Although we have projects where you write your own codes, we will also focus on two specific Python packages for Machine Learning, Scikit-Learn and -Tensorflow (see below for links etc). Moreover, the examples we +Tensorflow with Keras (see below for links etc). Moreover, the examples we introduce will serve as inputs to many of our discussions later, as well as allowing you to set up models and produce your own data and get started with programming. @@ -415,7 +399,7 @@ get started with programming.

    2684. 25
    2685. 26
    2686. ...
    2687. -
    2688. 66
    2689. +
    2690. 62
    2691. »
    2692. diff --git a/doc/pub/week34/html/._week34-bs017.html b/doc/pub/week34/html/._week34-bs017.html index ab6dee720..731a1423d 100644 --- a/doc/pub/week34/html/._week34-bs017.html +++ b/doc/pub/week34/html/._week34-bs017.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    2693. Installing R, C++, cython or Julia
    2694. Installing R, C++, cython, Numba etc
    2695. Numpy examples and Important Matrix and vector handling packages
    2696. -
    2697. Basic Matrix Features
    2698. -
    2699.    Some famous Matrices
    2700. -
    2701.    More Basic Matrix Features
    2702. -
    2703. Numpy and arrays
    2704. -
    2705. Matrices in Python
    2706. -
    2707. Meet the Pandas
    2708. -
    2709. Friday August 27
    2710. -
    2711.    Simple linear regression model using scikit-learn
    2712. -
    2713.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2714. -
    2715.    Organizing our data
    2716. -
    2717.    Seeing the wood for the trees
    2718. -
    2719.    And what about using neural networks?
    2720. -
    2721. A first summary
    2722. -
    2723. Why Linear Regression (aka Ordinary Least Squares and family)
    2724. -
    2725. Regression analysis, overarching aims
    2726. -
    2727. Regression analysis, overarching aims II
    2728. -
    2729. Examples
    2730. -
    2731. General linear models
    2732. -
    2733. Rewriting the fitting procedure as a linear algebra problem
    2734. -
    2735. Rewriting the fitting procedure as a linear algebra problem, more details
    2736. -
    2737. Generalizing the fitting procedure as a linear algebra problem
    2738. -
    2739. Generalizing the fitting procedure as a linear algebra problem
    2740. -
    2741. Optimizing our parameters
    2742. -
    2743. Our model for the nuclear binding energies
    2744. -
    2745. Optimizing our parameters, more details
    2746. -
    2747. Interpretations and optimizing our parameters
    2748. -
    2749. Interpretations and optimizing our parameters
    2750. -
    2751. Some useful matrix and vector expressions
    2752. -
    2753. Interpretations and optimizing our parameters
    2754. -
    2755. Own code for Ordinary Least Squares
    2756. -
    2757. Adding error analysis and training set up
    2758. -
    2759. The \( \chi^2 \) function
    2760. -
    2761. The \( \chi^2 \) function
    2762. -
    2763. The \( \chi^2 \) function
    2764. -
    2765. The \( \chi^2 \) function
    2766. -
    2767. The \( \chi^2 \) function
    2768. -
    2769. The \( \chi^2 \) function
    2770. -
    2771. Fitting an Equation of State for Dense Nuclear Matter
    2772. -
    2773. The code
    2774. -
    2775. Splitting our Data in Training and Test data
    2776. -
    2777. Exercises
    2778. -
    2779. Exercise 1: Setting up various Python environments
    2780. -
    2781. Exercise 2: making your own data and exploring scikit-learn
    2782. -
    2783. Exercise 3: Normalizing our data
    2784. +
    2785. Numpy and arrays
    2786. +
    2787. Matrices in Python
    2788. +
    2789. Meet the Pandas
    2790. +
    2791.    Simple linear regression model using scikit-learn
    2792. +
    2793.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2794. +
    2795.    Organizing our data
    2796. +
    2797.    And what about using neural networks?
    2798. +
    2799. A first summary
    2800. +
    2801. Why Linear Regression (aka Ordinary Least Squares and family)
    2802. +
    2803. Regression analysis, overarching aims
    2804. +
    2805. Regression analysis, overarching aims II
    2806. +
    2807. Examples
    2808. +
    2809. General linear models
    2810. +
    2811. Rewriting the fitting procedure as a linear algebra problem
    2812. +
    2813. Rewriting the fitting procedure as a linear algebra problem, more details
    2814. +
    2815. Generalizing the fitting procedure as a linear algebra problem
    2816. +
    2817. Generalizing the fitting procedure as a linear algebra problem
    2818. +
    2819. Optimizing our parameters
    2820. +
    2821. Our model for the nuclear binding energies
    2822. +
    2823. Optimizing our parameters, more details
    2824. +
    2825. Interpretations and optimizing our parameters
    2826. +
    2827. Interpretations and optimizing our parameters
    2828. +
    2829. Some useful matrix and vector expressions
    2830. +
    2831. Interpretations and optimizing our parameters
    2832. +
    2833. Own code for Ordinary Least Squares
    2834. +
    2835. Adding error analysis and training set up
    2836. +
    2837. The \( \chi^2 \) function
    2838. +
    2839. The \( \chi^2 \) function
    2840. +
    2841. The \( \chi^2 \) function
    2842. +
    2843. The \( \chi^2 \) function
    2844. +
    2845. The \( \chi^2 \) function
    2846. +
    2847. The \( \chi^2 \) function
    2848. +
    2849. Fitting an Equation of State for Dense Nuclear Matter
    2850. +
    2851. The code
    2852. +
    2853. Splitting our Data in Training and Test data
    2854. +
    2855. Exercises
    2856. +
    2857. Exercise 1: Setting up various Python environments
    2858. +
    2859. Exercise 2: making your own data and exploring scikit-learn
    2860. +
    2861. Exercise 3: Split data in test and training data
    2862. @@ -446,7 +430,7 @@ of algorithms and methods we will discuss.
    2863. 26
    2864. 27
    2865. ...
    2866. -
    2867. 66
    2868. +
    2869. 62
    2870. »
    2871. diff --git a/doc/pub/week34/html/._week34-bs018.html b/doc/pub/week34/html/._week34-bs018.html index 6ba6a4a4c..0fc707830 100644 --- a/doc/pub/week34/html/._week34-bs018.html +++ b/doc/pub/week34/html/._week34-bs018.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    2872. Installing R, C++, cython or Julia
    2873. Installing R, C++, cython, Numba etc
    2874. Numpy examples and Important Matrix and vector handling packages
    2875. -
    2876. Basic Matrix Features
    2877. -
    2878.    Some famous Matrices
    2879. -
    2880.    More Basic Matrix Features
    2881. -
    2882. Numpy and arrays
    2883. -
    2884. Matrices in Python
    2885. -
    2886. Meet the Pandas
    2887. -
    2888. Friday August 27
    2889. -
    2890.    Simple linear regression model using scikit-learn
    2891. -
    2892.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2893. -
    2894.    Organizing our data
    2895. -
    2896.    Seeing the wood for the trees
    2897. -
    2898.    And what about using neural networks?
    2899. -
    2900. A first summary
    2901. -
    2902. Why Linear Regression (aka Ordinary Least Squares and family)
    2903. -
    2904. Regression analysis, overarching aims
    2905. -
    2906. Regression analysis, overarching aims II
    2907. -
    2908. Examples
    2909. -
    2910. General linear models
    2911. -
    2912. Rewriting the fitting procedure as a linear algebra problem
    2913. -
    2914. Rewriting the fitting procedure as a linear algebra problem, more details
    2915. -
    2916. Generalizing the fitting procedure as a linear algebra problem
    2917. -
    2918. Generalizing the fitting procedure as a linear algebra problem
    2919. -
    2920. Optimizing our parameters
    2921. -
    2922. Our model for the nuclear binding energies
    2923. -
    2924. Optimizing our parameters, more details
    2925. -
    2926. Interpretations and optimizing our parameters
    2927. -
    2928. Interpretations and optimizing our parameters
    2929. -
    2930. Some useful matrix and vector expressions
    2931. -
    2932. Interpretations and optimizing our parameters
    2933. -
    2934. Own code for Ordinary Least Squares
    2935. -
    2936. Adding error analysis and training set up
    2937. -
    2938. The \( \chi^2 \) function
    2939. -
    2940. The \( \chi^2 \) function
    2941. -
    2942. The \( \chi^2 \) function
    2943. -
    2944. The \( \chi^2 \) function
    2945. -
    2946. The \( \chi^2 \) function
    2947. -
    2948. The \( \chi^2 \) function
    2949. -
    2950. Fitting an Equation of State for Dense Nuclear Matter
    2951. -
    2952. The code
    2953. -
    2954. Splitting our Data in Training and Test data
    2955. -
    2956. Exercises
    2957. -
    2958. Exercise 1: Setting up various Python environments
    2959. -
    2960. Exercise 2: making your own data and exploring scikit-learn
    2961. -
    2962. Exercise 3: Normalizing our data
    2963. +
    2964. Numpy and arrays
    2965. +
    2966. Matrices in Python
    2967. +
    2968. Meet the Pandas
    2969. +
    2970.    Simple linear regression model using scikit-learn
    2971. +
    2972.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    2973. +
    2974.    Organizing our data
    2975. +
    2976.    And what about using neural networks?
    2977. +
    2978. A first summary
    2979. +
    2980. Why Linear Regression (aka Ordinary Least Squares and family)
    2981. +
    2982. Regression analysis, overarching aims
    2983. +
    2984. Regression analysis, overarching aims II
    2985. +
    2986. Examples
    2987. +
    2988. General linear models
    2989. +
    2990. Rewriting the fitting procedure as a linear algebra problem
    2991. +
    2992. Rewriting the fitting procedure as a linear algebra problem, more details
    2993. +
    2994. Generalizing the fitting procedure as a linear algebra problem
    2995. +
    2996. Generalizing the fitting procedure as a linear algebra problem
    2997. +
    2998. Optimizing our parameters
    2999. +
    3000. Our model for the nuclear binding energies
    3001. +
    3002. Optimizing our parameters, more details
    3003. +
    3004. Interpretations and optimizing our parameters
    3005. +
    3006. Interpretations and optimizing our parameters
    3007. +
    3008. Some useful matrix and vector expressions
    3009. +
    3010. Interpretations and optimizing our parameters
    3011. +
    3012. Own code for Ordinary Least Squares
    3013. +
    3014. Adding error analysis and training set up
    3015. +
    3016. The \( \chi^2 \) function
    3017. +
    3018. The \( \chi^2 \) function
    3019. +
    3020. The \( \chi^2 \) function
    3021. +
    3022. The \( \chi^2 \) function
    3023. +
    3024. The \( \chi^2 \) function
    3025. +
    3026. The \( \chi^2 \) function
    3027. +
    3028. Fitting an Equation of State for Dense Nuclear Matter
    3029. +
    3030. The code
    3031. +
    3032. Splitting our Data in Training and Test data
    3033. +
    3034. Exercises
    3035. +
    3036. Exercise 1: Setting up various Python environments
    3037. +
    3038. Exercise 2: making your own data and exploring scikit-learn
    3039. +
    3040. Exercise 3: Split data in test and training data
    3041. @@ -398,7 +382,7 @@ desired output of a system. Some of the most common tasks are:
    3042. 27
    3043. 28
    3044. ...
    3045. -
    3046. 66
    3047. +
    3048. 62
    3049. »
    3050. diff --git a/doc/pub/week34/html/._week34-bs019.html b/doc/pub/week34/html/._week34-bs019.html index 0f1c39c1b..ecb4e968d 100644 --- a/doc/pub/week34/html/._week34-bs019.html +++ b/doc/pub/week34/html/._week34-bs019.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    3051. Installing R, C++, cython or Julia
    3052. Installing R, C++, cython, Numba etc
    3053. Numpy examples and Important Matrix and vector handling packages
    3054. -
    3055. Basic Matrix Features
    3056. -
    3057.    Some famous Matrices
    3058. -
    3059.    More Basic Matrix Features
    3060. -
    3061. Numpy and arrays
    3062. -
    3063. Matrices in Python
    3064. -
    3065. Meet the Pandas
    3066. -
    3067. Friday August 27
    3068. -
    3069.    Simple linear regression model using scikit-learn
    3070. -
    3071.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3072. -
    3073.    Organizing our data
    3074. -
    3075.    Seeing the wood for the trees
    3076. -
    3077.    And what about using neural networks?
    3078. -
    3079. A first summary
    3080. -
    3081. Why Linear Regression (aka Ordinary Least Squares and family)
    3082. -
    3083. Regression analysis, overarching aims
    3084. -
    3085. Regression analysis, overarching aims II
    3086. -
    3087. Examples
    3088. -
    3089. General linear models
    3090. -
    3091. Rewriting the fitting procedure as a linear algebra problem
    3092. -
    3093. Rewriting the fitting procedure as a linear algebra problem, more details
    3094. -
    3095. Generalizing the fitting procedure as a linear algebra problem
    3096. -
    3097. Generalizing the fitting procedure as a linear algebra problem
    3098. -
    3099. Optimizing our parameters
    3100. -
    3101. Our model for the nuclear binding energies
    3102. -
    3103. Optimizing our parameters, more details
    3104. -
    3105. Interpretations and optimizing our parameters
    3106. -
    3107. Interpretations and optimizing our parameters
    3108. -
    3109. Some useful matrix and vector expressions
    3110. -
    3111. Interpretations and optimizing our parameters
    3112. -
    3113. Own code for Ordinary Least Squares
    3114. -
    3115. Adding error analysis and training set up
    3116. -
    3117. The \( \chi^2 \) function
    3118. -
    3119. The \( \chi^2 \) function
    3120. -
    3121. The \( \chi^2 \) function
    3122. -
    3123. The \( \chi^2 \) function
    3124. -
    3125. The \( \chi^2 \) function
    3126. -
    3127. The \( \chi^2 \) function
    3128. -
    3129. Fitting an Equation of State for Dense Nuclear Matter
    3130. -
    3131. The code
    3132. -
    3133. Splitting our Data in Training and Test data
    3134. -
    3135. Exercises
    3136. -
    3137. Exercise 1: Setting up various Python environments
    3138. -
    3139. Exercise 2: making your own data and exploring scikit-learn
    3140. -
    3141. Exercise 3: Normalizing our data
    3142. +
    3143. Numpy and arrays
    3144. +
    3145. Matrices in Python
    3146. +
    3147. Meet the Pandas
    3148. +
    3149.    Simple linear regression model using scikit-learn
    3150. +
    3151.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3152. +
    3153.    Organizing our data
    3154. +
    3155.    And what about using neural networks?
    3156. +
    3157. A first summary
    3158. +
    3159. Why Linear Regression (aka Ordinary Least Squares and family)
    3160. +
    3161. Regression analysis, overarching aims
    3162. +
    3163. Regression analysis, overarching aims II
    3164. +
    3165. Examples
    3166. +
    3167. General linear models
    3168. +
    3169. Rewriting the fitting procedure as a linear algebra problem
    3170. +
    3171. Rewriting the fitting procedure as a linear algebra problem, more details
    3172. +
    3173. Generalizing the fitting procedure as a linear algebra problem
    3174. +
    3175. Generalizing the fitting procedure as a linear algebra problem
    3176. +
    3177. Optimizing our parameters
    3178. +
    3179. Our model for the nuclear binding energies
    3180. +
    3181. Optimizing our parameters, more details
    3182. +
    3183. Interpretations and optimizing our parameters
    3184. +
    3185. Interpretations and optimizing our parameters
    3186. +
    3187. Some useful matrix and vector expressions
    3188. +
    3189. Interpretations and optimizing our parameters
    3190. +
    3191. Own code for Ordinary Least Squares
    3192. +
    3193. Adding error analysis and training set up
    3194. +
    3195. The \( \chi^2 \) function
    3196. +
    3197. The \( \chi^2 \) function
    3198. +
    3199. The \( \chi^2 \) function
    3200. +
    3201. The \( \chi^2 \) function
    3202. +
    3203. The \( \chi^2 \) function
    3204. +
    3205. The \( \chi^2 \) function
    3206. +
    3207. Fitting an Equation of State for Dense Nuclear Matter
    3208. +
    3209. The code
    3210. +
    3211. Splitting our Data in Training and Test data
    3212. +
    3213. Exercises
    3214. +
    3215. Exercise 1: Setting up various Python environments
    3216. +
    3217. Exercise 2: making your own data and exploring scikit-learn
    3218. +
    3219. Exercise 3: Split data in test and training data
    3220. @@ -389,7 +373,7 @@ whether we deal with supervised or unsupervised learning.
    3221. 28
    3222. 29
    3223. ...
    3224. -
    3225. 66
    3226. +
    3227. 62
    3228. »
    3229. diff --git a/doc/pub/week34/html/._week34-bs020.html b/doc/pub/week34/html/._week34-bs020.html index c3d562647..bbcbcae60 100644 --- a/doc/pub/week34/html/._week34-bs020.html +++ b/doc/pub/week34/html/._week34-bs020.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    3230. Installing R, C++, cython or Julia
    3231. Installing R, C++, cython, Numba etc
    3232. Numpy examples and Important Matrix and vector handling packages
    3233. -
    3234. Basic Matrix Features
    3235. -
    3236.    Some famous Matrices
    3237. -
    3238.    More Basic Matrix Features
    3239. -
    3240. Numpy and arrays
    3241. -
    3242. Matrices in Python
    3243. -
    3244. Meet the Pandas
    3245. -
    3246. Friday August 27
    3247. -
    3248.    Simple linear regression model using scikit-learn
    3249. -
    3250.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3251. -
    3252.    Organizing our data
    3253. -
    3254.    Seeing the wood for the trees
    3255. -
    3256.    And what about using neural networks?
    3257. -
    3258. A first summary
    3259. -
    3260. Why Linear Regression (aka Ordinary Least Squares and family)
    3261. -
    3262. Regression analysis, overarching aims
    3263. -
    3264. Regression analysis, overarching aims II
    3265. -
    3266. Examples
    3267. -
    3268. General linear models
    3269. -
    3270. Rewriting the fitting procedure as a linear algebra problem
    3271. -
    3272. Rewriting the fitting procedure as a linear algebra problem, more details
    3273. -
    3274. Generalizing the fitting procedure as a linear algebra problem
    3275. -
    3276. Generalizing the fitting procedure as a linear algebra problem
    3277. -
    3278. Optimizing our parameters
    3279. -
    3280. Our model for the nuclear binding energies
    3281. -
    3282. Optimizing our parameters, more details
    3283. -
    3284. Interpretations and optimizing our parameters
    3285. -
    3286. Interpretations and optimizing our parameters
    3287. -
    3288. Some useful matrix and vector expressions
    3289. -
    3290. Interpretations and optimizing our parameters
    3291. -
    3292. Own code for Ordinary Least Squares
    3293. -
    3294. Adding error analysis and training set up
    3295. -
    3296. The \( \chi^2 \) function
    3297. -
    3298. The \( \chi^2 \) function
    3299. -
    3300. The \( \chi^2 \) function
    3301. -
    3302. The \( \chi^2 \) function
    3303. -
    3304. The \( \chi^2 \) function
    3305. -
    3306. The \( \chi^2 \) function
    3307. -
    3308. Fitting an Equation of State for Dense Nuclear Matter
    3309. -
    3310. The code
    3311. -
    3312. Splitting our Data in Training and Test data
    3313. -
    3314. Exercises
    3315. -
    3316. Exercise 1: Setting up various Python environments
    3317. -
    3318. Exercise 2: making your own data and exploring scikit-learn
    3319. -
    3320. Exercise 3: Normalizing our data
    3321. +
    3322. Numpy and arrays
    3323. +
    3324. Matrices in Python
    3325. +
    3326. Meet the Pandas
    3327. +
    3328.    Simple linear regression model using scikit-learn
    3329. +
    3330.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3331. +
    3332.    Organizing our data
    3333. +
    3334.    And what about using neural networks?
    3335. +
    3336. A first summary
    3337. +
    3338. Why Linear Regression (aka Ordinary Least Squares and family)
    3339. +
    3340. Regression analysis, overarching aims
    3341. +
    3342. Regression analysis, overarching aims II
    3343. +
    3344. Examples
    3345. +
    3346. General linear models
    3347. +
    3348. Rewriting the fitting procedure as a linear algebra problem
    3349. +
    3350. Rewriting the fitting procedure as a linear algebra problem, more details
    3351. +
    3352. Generalizing the fitting procedure as a linear algebra problem
    3353. +
    3354. Generalizing the fitting procedure as a linear algebra problem
    3355. +
    3356. Optimizing our parameters
    3357. +
    3358. Our model for the nuclear binding energies
    3359. +
    3360. Optimizing our parameters, more details
    3361. +
    3362. Interpretations and optimizing our parameters
    3363. +
    3364. Interpretations and optimizing our parameters
    3365. +
    3366. Some useful matrix and vector expressions
    3367. +
    3368. Interpretations and optimizing our parameters
    3369. +
    3370. Own code for Ordinary Least Squares
    3371. +
    3372. Adding error analysis and training set up
    3373. +
    3374. The \( \chi^2 \) function
    3375. +
    3376. The \( \chi^2 \) function
    3377. +
    3378. The \( \chi^2 \) function
    3379. +
    3380. The \( \chi^2 \) function
    3381. +
    3382. The \( \chi^2 \) function
    3383. +
    3384. The \( \chi^2 \) function
    3385. +
    3386. Fitting an Equation of State for Dense Nuclear Matter
    3387. +
    3388. The code
    3389. +
    3390. Splitting our Data in Training and Test data
    3391. +
    3392. Exercises
    3393. +
    3394. Exercise 1: Setting up various Python environments
    3395. +
    3396. Exercise 2: making your own data and exploring scikit-learn
    3397. +
    3398. Exercise 3: Split data in test and training data
    3399. @@ -380,7 +364,7 @@ MathJax.Hub.Config({
    3400. 29
    3401. 30
    3402. ...
    3403. -
    3404. 66
    3405. +
    3406. 62
    3407. »
    3408. diff --git a/doc/pub/week34/html/._week34-bs021.html b/doc/pub/week34/html/._week34-bs021.html index 09005b222..f7e277367 100644 --- a/doc/pub/week34/html/._week34-bs021.html +++ b/doc/pub/week34/html/._week34-bs021.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    3409. Installing R, C++, cython or Julia
    3410. Installing R, C++, cython, Numba etc
    3411. Numpy examples and Important Matrix and vector handling packages
    3412. -
    3413. Basic Matrix Features
    3414. -
    3415.    Some famous Matrices
    3416. -
    3417.    More Basic Matrix Features
    3418. -
    3419. Numpy and arrays
    3420. -
    3421. Matrices in Python
    3422. -
    3423. Meet the Pandas
    3424. -
    3425. Friday August 27
    3426. -
    3427.    Simple linear regression model using scikit-learn
    3428. -
    3429.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3430. -
    3431.    Organizing our data
    3432. -
    3433.    Seeing the wood for the trees
    3434. -
    3435.    And what about using neural networks?
    3436. -
    3437. A first summary
    3438. -
    3439. Why Linear Regression (aka Ordinary Least Squares and family)
    3440. -
    3441. Regression analysis, overarching aims
    3442. -
    3443. Regression analysis, overarching aims II
    3444. -
    3445. Examples
    3446. -
    3447. General linear models
    3448. -
    3449. Rewriting the fitting procedure as a linear algebra problem
    3450. -
    3451. Rewriting the fitting procedure as a linear algebra problem, more details
    3452. -
    3453. Generalizing the fitting procedure as a linear algebra problem
    3454. -
    3455. Generalizing the fitting procedure as a linear algebra problem
    3456. -
    3457. Optimizing our parameters
    3458. -
    3459. Our model for the nuclear binding energies
    3460. -
    3461. Optimizing our parameters, more details
    3462. -
    3463. Interpretations and optimizing our parameters
    3464. -
    3465. Interpretations and optimizing our parameters
    3466. -
    3467. Some useful matrix and vector expressions
    3468. -
    3469. Interpretations and optimizing our parameters
    3470. -
    3471. Own code for Ordinary Least Squares
    3472. -
    3473. Adding error analysis and training set up
    3474. -
    3475. The \( \chi^2 \) function
    3476. -
    3477. The \( \chi^2 \) function
    3478. -
    3479. The \( \chi^2 \) function
    3480. -
    3481. The \( \chi^2 \) function
    3482. -
    3483. The \( \chi^2 \) function
    3484. -
    3485. The \( \chi^2 \) function
    3486. -
    3487. Fitting an Equation of State for Dense Nuclear Matter
    3488. -
    3489. The code
    3490. -
    3491. Splitting our Data in Training and Test data
    3492. -
    3493. Exercises
    3494. -
    3495. Exercise 1: Setting up various Python environments
    3496. -
    3497. Exercise 2: making your own data and exploring scikit-learn
    3498. -
    3499. Exercise 3: Normalizing our data
    3500. +
    3501. Numpy and arrays
    3502. +
    3503. Matrices in Python
    3504. +
    3505. Meet the Pandas
    3506. +
    3507.    Simple linear regression model using scikit-learn
    3508. +
    3509.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3510. +
    3511.    Organizing our data
    3512. +
    3513.    And what about using neural networks?
    3514. +
    3515. A first summary
    3516. +
    3517. Why Linear Regression (aka Ordinary Least Squares and family)
    3518. +
    3519. Regression analysis, overarching aims
    3520. +
    3521. Regression analysis, overarching aims II
    3522. +
    3523. Examples
    3524. +
    3525. General linear models
    3526. +
    3527. Rewriting the fitting procedure as a linear algebra problem
    3528. +
    3529. Rewriting the fitting procedure as a linear algebra problem, more details
    3530. +
    3531. Generalizing the fitting procedure as a linear algebra problem
    3532. +
    3533. Generalizing the fitting procedure as a linear algebra problem
    3534. +
    3535. Optimizing our parameters
    3536. +
    3537. Our model for the nuclear binding energies
    3538. +
    3539. Optimizing our parameters, more details
    3540. +
    3541. Interpretations and optimizing our parameters
    3542. +
    3543. Interpretations and optimizing our parameters
    3544. +
    3545. Some useful matrix and vector expressions
    3546. +
    3547. Interpretations and optimizing our parameters
    3548. +
    3549. Own code for Ordinary Least Squares
    3550. +
    3551. Adding error analysis and training set up
    3552. +
    3553. The \( \chi^2 \) function
    3554. +
    3555. The \( \chi^2 \) function
    3556. +
    3557. The \( \chi^2 \) function
    3558. +
    3559. The \( \chi^2 \) function
    3560. +
    3561. The \( \chi^2 \) function
    3562. +
    3563. The \( \chi^2 \) function
    3564. +
    3565. Fitting an Equation of State for Dense Nuclear Matter
    3566. +
    3567. The code
    3568. +
    3569. Splitting our Data in Training and Test data
    3570. +
    3571. Exercises
    3572. +
    3573. Exercise 1: Setting up various Python environments
    3574. +
    3575. Exercise 2: making your own data and exploring scikit-learn
    3576. +
    3577. Exercise 3: Split data in test and training data
    3578. @@ -406,7 +390,7 @@ what is the likelihood of finding \( B \).
    3579. 30
    3580. 31
    3581. ...
    3582. -
    3583. 66
    3584. +
    3585. 62
    3586. »
    3587. diff --git a/doc/pub/week34/html/._week34-bs022.html b/doc/pub/week34/html/._week34-bs022.html index e42624982..79bcf29c2 100644 --- a/doc/pub/week34/html/._week34-bs022.html +++ b/doc/pub/week34/html/._week34-bs022.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    3588. Installing R, C++, cython or Julia
    3589. Installing R, C++, cython, Numba etc
    3590. Numpy examples and Important Matrix and vector handling packages
    3591. -
    3592. Basic Matrix Features
    3593. -
    3594.    Some famous Matrices
    3595. -
    3596.    More Basic Matrix Features
    3597. -
    3598. Numpy and arrays
    3599. -
    3600. Matrices in Python
    3601. -
    3602. Meet the Pandas
    3603. -
    3604. Friday August 27
    3605. -
    3606.    Simple linear regression model using scikit-learn
    3607. -
    3608.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3609. -
    3610.    Organizing our data
    3611. -
    3612.    Seeing the wood for the trees
    3613. -
    3614.    And what about using neural networks?
    3615. -
    3616. A first summary
    3617. -
    3618. Why Linear Regression (aka Ordinary Least Squares and family)
    3619. -
    3620. Regression analysis, overarching aims
    3621. -
    3622. Regression analysis, overarching aims II
    3623. -
    3624. Examples
    3625. -
    3626. General linear models
    3627. -
    3628. Rewriting the fitting procedure as a linear algebra problem
    3629. -
    3630. Rewriting the fitting procedure as a linear algebra problem, more details
    3631. -
    3632. Generalizing the fitting procedure as a linear algebra problem
    3633. -
    3634. Generalizing the fitting procedure as a linear algebra problem
    3635. -
    3636. Optimizing our parameters
    3637. -
    3638. Our model for the nuclear binding energies
    3639. -
    3640. Optimizing our parameters, more details
    3641. -
    3642. Interpretations and optimizing our parameters
    3643. -
    3644. Interpretations and optimizing our parameters
    3645. -
    3646. Some useful matrix and vector expressions
    3647. -
    3648. Interpretations and optimizing our parameters
    3649. -
    3650. Own code for Ordinary Least Squares
    3651. -
    3652. Adding error analysis and training set up
    3653. -
    3654. The \( \chi^2 \) function
    3655. -
    3656. The \( \chi^2 \) function
    3657. -
    3658. The \( \chi^2 \) function
    3659. -
    3660. The \( \chi^2 \) function
    3661. -
    3662. The \( \chi^2 \) function
    3663. -
    3664. The \( \chi^2 \) function
    3665. -
    3666. Fitting an Equation of State for Dense Nuclear Matter
    3667. -
    3668. The code
    3669. -
    3670. Splitting our Data in Training and Test data
    3671. -
    3672. Exercises
    3673. -
    3674. Exercise 1: Setting up various Python environments
    3675. -
    3676. Exercise 2: making your own data and exploring scikit-learn
    3677. -
    3678. Exercise 3: Normalizing our data
    3679. +
    3680. Numpy and arrays
    3681. +
    3682. Matrices in Python
    3683. +
    3684. Meet the Pandas
    3685. +
    3686.    Simple linear regression model using scikit-learn
    3687. +
    3688.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3689. +
    3690.    Organizing our data
    3691. +
    3692.    And what about using neural networks?
    3693. +
    3694. A first summary
    3695. +
    3696. Why Linear Regression (aka Ordinary Least Squares and family)
    3697. +
    3698. Regression analysis, overarching aims
    3699. +
    3700. Regression analysis, overarching aims II
    3701. +
    3702. Examples
    3703. +
    3704. General linear models
    3705. +
    3706. Rewriting the fitting procedure as a linear algebra problem
    3707. +
    3708. Rewriting the fitting procedure as a linear algebra problem, more details
    3709. +
    3710. Generalizing the fitting procedure as a linear algebra problem
    3711. +
    3712. Generalizing the fitting procedure as a linear algebra problem
    3713. +
    3714. Optimizing our parameters
    3715. +
    3716. Our model for the nuclear binding energies
    3717. +
    3718. Optimizing our parameters, more details
    3719. +
    3720. Interpretations and optimizing our parameters
    3721. +
    3722. Interpretations and optimizing our parameters
    3723. +
    3724. Some useful matrix and vector expressions
    3725. +
    3726. Interpretations and optimizing our parameters
    3727. +
    3728. Own code for Ordinary Least Squares
    3729. +
    3730. Adding error analysis and training set up
    3731. +
    3732. The \( \chi^2 \) function
    3733. +
    3734. The \( \chi^2 \) function
    3735. +
    3736. The \( \chi^2 \) function
    3737. +
    3738. The \( \chi^2 \) function
    3739. +
    3740. The \( \chi^2 \) function
    3741. +
    3742. The \( \chi^2 \) function
    3743. +
    3744. Fitting an Equation of State for Dense Nuclear Matter
    3745. +
    3746. The code
    3747. +
    3748. Splitting our Data in Training and Test data
    3749. +
    3750. Exercises
    3751. +
    3752. Exercise 1: Setting up various Python environments
    3753. +
    3754. Exercise 2: making your own data and exploring scikit-learn
    3755. +
    3756. Exercise 3: Split data in test and training data
    3757. @@ -403,7 +387,7 @@ could easily be many different models that fit the given data set equally we
    3758. 31
    3759. 32
    3760. ...
    3761. -
    3762. 66
    3763. +
    3764. 62
    3765. »
    3766. diff --git a/doc/pub/week34/html/._week34-bs023.html b/doc/pub/week34/html/._week34-bs023.html index 0ea4198fe..1549569bc 100644 --- a/doc/pub/week34/html/._week34-bs023.html +++ b/doc/pub/week34/html/._week34-bs023.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    3767. Installing R, C++, cython or Julia
    3768. Installing R, C++, cython, Numba etc
    3769. Numpy examples and Important Matrix and vector handling packages
    3770. -
    3771. Basic Matrix Features
    3772. -
    3773.    Some famous Matrices
    3774. -
    3775.    More Basic Matrix Features
    3776. -
    3777. Numpy and arrays
    3778. -
    3779. Matrices in Python
    3780. -
    3781. Meet the Pandas
    3782. -
    3783. Friday August 27
    3784. -
    3785.    Simple linear regression model using scikit-learn
    3786. -
    3787.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3788. -
    3789.    Organizing our data
    3790. -
    3791.    Seeing the wood for the trees
    3792. -
    3793.    And what about using neural networks?
    3794. -
    3795. A first summary
    3796. -
    3797. Why Linear Regression (aka Ordinary Least Squares and family)
    3798. -
    3799. Regression analysis, overarching aims
    3800. -
    3801. Regression analysis, overarching aims II
    3802. -
    3803. Examples
    3804. -
    3805. General linear models
    3806. -
    3807. Rewriting the fitting procedure as a linear algebra problem
    3808. -
    3809. Rewriting the fitting procedure as a linear algebra problem, more details
    3810. -
    3811. Generalizing the fitting procedure as a linear algebra problem
    3812. -
    3813. Generalizing the fitting procedure as a linear algebra problem
    3814. -
    3815. Optimizing our parameters
    3816. -
    3817. Our model for the nuclear binding energies
    3818. -
    3819. Optimizing our parameters, more details
    3820. -
    3821. Interpretations and optimizing our parameters
    3822. -
    3823. Interpretations and optimizing our parameters
    3824. -
    3825. Some useful matrix and vector expressions
    3826. -
    3827. Interpretations and optimizing our parameters
    3828. -
    3829. Own code for Ordinary Least Squares
    3830. -
    3831. Adding error analysis and training set up
    3832. -
    3833. The \( \chi^2 \) function
    3834. -
    3835. The \( \chi^2 \) function
    3836. -
    3837. The \( \chi^2 \) function
    3838. -
    3839. The \( \chi^2 \) function
    3840. -
    3841. The \( \chi^2 \) function
    3842. -
    3843. The \( \chi^2 \) function
    3844. -
    3845. Fitting an Equation of State for Dense Nuclear Matter
    3846. -
    3847. The code
    3848. -
    3849. Splitting our Data in Training and Test data
    3850. -
    3851. Exercises
    3852. -
    3853. Exercise 1: Setting up various Python environments
    3854. -
    3855. Exercise 2: making your own data and exploring scikit-learn
    3856. -
    3857. Exercise 3: Normalizing our data
    3858. +
    3859. Numpy and arrays
    3860. +
    3861. Matrices in Python
    3862. +
    3863. Meet the Pandas
    3864. +
    3865.    Simple linear regression model using scikit-learn
    3866. +
    3867.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3868. +
    3869.    Organizing our data
    3870. +
    3871.    And what about using neural networks?
    3872. +
    3873. A first summary
    3874. +
    3875. Why Linear Regression (aka Ordinary Least Squares and family)
    3876. +
    3877. Regression analysis, overarching aims
    3878. +
    3879. Regression analysis, overarching aims II
    3880. +
    3881. Examples
    3882. +
    3883. General linear models
    3884. +
    3885. Rewriting the fitting procedure as a linear algebra problem
    3886. +
    3887. Rewriting the fitting procedure as a linear algebra problem, more details
    3888. +
    3889. Generalizing the fitting procedure as a linear algebra problem
    3890. +
    3891. Generalizing the fitting procedure as a linear algebra problem
    3892. +
    3893. Optimizing our parameters
    3894. +
    3895. Our model for the nuclear binding energies
    3896. +
    3897. Optimizing our parameters, more details
    3898. +
    3899. Interpretations and optimizing our parameters
    3900. +
    3901. Interpretations and optimizing our parameters
    3902. +
    3903. Some useful matrix and vector expressions
    3904. +
    3905. Interpretations and optimizing our parameters
    3906. +
    3907. Own code for Ordinary Least Squares
    3908. +
    3909. Adding error analysis and training set up
    3910. +
    3911. The \( \chi^2 \) function
    3912. +
    3913. The \( \chi^2 \) function
    3914. +
    3915. The \( \chi^2 \) function
    3916. +
    3917. The \( \chi^2 \) function
    3918. +
    3919. The \( \chi^2 \) function
    3920. +
    3921. The \( \chi^2 \) function
    3922. +
    3923. Fitting an Equation of State for Dense Nuclear Matter
    3924. +
    3925. The code
    3926. +
    3927. Splitting our Data in Training and Test data
    3928. +
    3929. Exercises
    3930. +
    3931. Exercise 1: Setting up various Python environments
    3932. +
    3933. Exercise 2: making your own data and exploring scikit-learn
    3934. +
    3935. Exercise 3: Split data in test and training data
    3936. @@ -401,7 +385,7 @@ may first try the simplest class of models, namely linear models, followed obvio
    3937. 32
    3938. 33
    3939. ...
    3940. -
    3941. 66
    3942. +
    3943. 62
    3944. »
    3945. diff --git a/doc/pub/week34/html/._week34-bs024.html b/doc/pub/week34/html/._week34-bs024.html index 163cccaeb..8a8526070 100644 --- a/doc/pub/week34/html/._week34-bs024.html +++ b/doc/pub/week34/html/._week34-bs024.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    3946. Installing R, C++, cython or Julia
    3947. Installing R, C++, cython, Numba etc
    3948. Numpy examples and Important Matrix and vector handling packages
    3949. -
    3950. Basic Matrix Features
    3951. -
    3952.    Some famous Matrices
    3953. -
    3954.    More Basic Matrix Features
    3955. -
    3956. Numpy and arrays
    3957. -
    3958. Matrices in Python
    3959. -
    3960. Meet the Pandas
    3961. -
    3962. Friday August 27
    3963. -
    3964.    Simple linear regression model using scikit-learn
    3965. -
    3966.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    3967. -
    3968.    Organizing our data
    3969. -
    3970.    Seeing the wood for the trees
    3971. -
    3972.    And what about using neural networks?
    3973. -
    3974. A first summary
    3975. -
    3976. Why Linear Regression (aka Ordinary Least Squares and family)
    3977. -
    3978. Regression analysis, overarching aims
    3979. -
    3980. Regression analysis, overarching aims II
    3981. -
    3982. Examples
    3983. -
    3984. General linear models
    3985. -
    3986. Rewriting the fitting procedure as a linear algebra problem
    3987. -
    3988. Rewriting the fitting procedure as a linear algebra problem, more details
    3989. -
    3990. Generalizing the fitting procedure as a linear algebra problem
    3991. -
    3992. Generalizing the fitting procedure as a linear algebra problem
    3993. -
    3994. Optimizing our parameters
    3995. -
    3996. Our model for the nuclear binding energies
    3997. -
    3998. Optimizing our parameters, more details
    3999. -
    4000. Interpretations and optimizing our parameters
    4001. -
    4002. Interpretations and optimizing our parameters
    4003. -
    4004. Some useful matrix and vector expressions
    4005. -
    4006. Interpretations and optimizing our parameters
    4007. -
    4008. Own code for Ordinary Least Squares
    4009. -
    4010. Adding error analysis and training set up
    4011. -
    4012. The \( \chi^2 \) function
    4013. -
    4014. The \( \chi^2 \) function
    4015. -
    4016. The \( \chi^2 \) function
    4017. -
    4018. The \( \chi^2 \) function
    4019. -
    4020. The \( \chi^2 \) function
    4021. -
    4022. The \( \chi^2 \) function
    4023. -
    4024. Fitting an Equation of State for Dense Nuclear Matter
    4025. -
    4026. The code
    4027. -
    4028. Splitting our Data in Training and Test data
    4029. -
    4030. Exercises
    4031. -
    4032. Exercise 1: Setting up various Python environments
    4033. -
    4034. Exercise 2: making your own data and exploring scikit-learn
    4035. -
    4036. Exercise 3: Normalizing our data
    4037. +
    4038. Numpy and arrays
    4039. +
    4040. Matrices in Python
    4041. +
    4042. Meet the Pandas
    4043. +
    4044.    Simple linear regression model using scikit-learn
    4045. +
    4046.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4047. +
    4048.    Organizing our data
    4049. +
    4050.    And what about using neural networks?
    4051. +
    4052. A first summary
    4053. +
    4054. Why Linear Regression (aka Ordinary Least Squares and family)
    4055. +
    4056. Regression analysis, overarching aims
    4057. +
    4058. Regression analysis, overarching aims II
    4059. +
    4060. Examples
    4061. +
    4062. General linear models
    4063. +
    4064. Rewriting the fitting procedure as a linear algebra problem
    4065. +
    4066. Rewriting the fitting procedure as a linear algebra problem, more details
    4067. +
    4068. Generalizing the fitting procedure as a linear algebra problem
    4069. +
    4070. Generalizing the fitting procedure as a linear algebra problem
    4071. +
    4072. Optimizing our parameters
    4073. +
    4074. Our model for the nuclear binding energies
    4075. +
    4076. Optimizing our parameters, more details
    4077. +
    4078. Interpretations and optimizing our parameters
    4079. +
    4080. Interpretations and optimizing our parameters
    4081. +
    4082. Some useful matrix and vector expressions
    4083. +
    4084. Interpretations and optimizing our parameters
    4085. +
    4086. Own code for Ordinary Least Squares
    4087. +
    4088. Adding error analysis and training set up
    4089. +
    4090. The \( \chi^2 \) function
    4091. +
    4092. The \( \chi^2 \) function
    4093. +
    4094. The \( \chi^2 \) function
    4095. +
    4096. The \( \chi^2 \) function
    4097. +
    4098. The \( \chi^2 \) function
    4099. +
    4100. The \( \chi^2 \) function
    4101. +
    4102. Fitting an Equation of State for Dense Nuclear Matter
    4103. +
    4104. The code
    4105. +
    4106. Splitting our Data in Training and Test data
    4107. +
    4108. Exercises
    4109. +
    4110. Exercise 1: Setting up various Python environments
    4111. +
    4112. Exercise 2: making your own data and exploring scikit-learn
    4113. +
    4114. Exercise 3: Split data in test and training data
    4115. @@ -414,7 +398,7 @@ you can use pip as well and simply install Python as
    4116. 33
    4117. 34
    4118. ...
    4119. -
    4120. 66
    4121. +
    4122. 62
    4123. »
    4124. diff --git a/doc/pub/week34/html/._week34-bs025.html b/doc/pub/week34/html/._week34-bs025.html index 0cedc4673..f25a18b9e 100644 --- a/doc/pub/week34/html/._week34-bs025.html +++ b/doc/pub/week34/html/._week34-bs025.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    4125. Installing R, C++, cython or Julia
    4126. Installing R, C++, cython, Numba etc
    4127. Numpy examples and Important Matrix and vector handling packages
    4128. -
    4129. Basic Matrix Features
    4130. -
    4131.    Some famous Matrices
    4132. -
    4133.    More Basic Matrix Features
    4134. -
    4135. Numpy and arrays
    4136. -
    4137. Matrices in Python
    4138. -
    4139. Meet the Pandas
    4140. -
    4141. Friday August 27
    4142. -
    4143.    Simple linear regression model using scikit-learn
    4144. -
    4145.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4146. -
    4147.    Organizing our data
    4148. -
    4149.    Seeing the wood for the trees
    4150. -
    4151.    And what about using neural networks?
    4152. -
    4153. A first summary
    4154. -
    4155. Why Linear Regression (aka Ordinary Least Squares and family)
    4156. -
    4157. Regression analysis, overarching aims
    4158. -
    4159. Regression analysis, overarching aims II
    4160. -
    4161. Examples
    4162. -
    4163. General linear models
    4164. -
    4165. Rewriting the fitting procedure as a linear algebra problem
    4166. -
    4167. Rewriting the fitting procedure as a linear algebra problem, more details
    4168. -
    4169. Generalizing the fitting procedure as a linear algebra problem
    4170. -
    4171. Generalizing the fitting procedure as a linear algebra problem
    4172. -
    4173. Optimizing our parameters
    4174. -
    4175. Our model for the nuclear binding energies
    4176. -
    4177. Optimizing our parameters, more details
    4178. -
    4179. Interpretations and optimizing our parameters
    4180. -
    4181. Interpretations and optimizing our parameters
    4182. -
    4183. Some useful matrix and vector expressions
    4184. -
    4185. Interpretations and optimizing our parameters
    4186. -
    4187. Own code for Ordinary Least Squares
    4188. -
    4189. Adding error analysis and training set up
    4190. -
    4191. The \( \chi^2 \) function
    4192. -
    4193. The \( \chi^2 \) function
    4194. -
    4195. The \( \chi^2 \) function
    4196. -
    4197. The \( \chi^2 \) function
    4198. -
    4199. The \( \chi^2 \) function
    4200. -
    4201. The \( \chi^2 \) function
    4202. -
    4203. Fitting an Equation of State for Dense Nuclear Matter
    4204. -
    4205. The code
    4206. -
    4207. Splitting our Data in Training and Test data
    4208. -
    4209. Exercises
    4210. -
    4211. Exercise 1: Setting up various Python environments
    4212. -
    4213. Exercise 2: making your own data and exploring scikit-learn
    4214. -
    4215. Exercise 3: Normalizing our data
    4216. +
    4217. Numpy and arrays
    4218. +
    4219. Matrices in Python
    4220. +
    4221. Meet the Pandas
    4222. +
    4223.    Simple linear regression model using scikit-learn
    4224. +
    4225.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4226. +
    4227.    Organizing our data
    4228. +
    4229.    And what about using neural networks?
    4230. +
    4231. A first summary
    4232. +
    4233. Why Linear Regression (aka Ordinary Least Squares and family)
    4234. +
    4235. Regression analysis, overarching aims
    4236. +
    4237. Regression analysis, overarching aims II
    4238. +
    4239. Examples
    4240. +
    4241. General linear models
    4242. +
    4243. Rewriting the fitting procedure as a linear algebra problem
    4244. +
    4245. Rewriting the fitting procedure as a linear algebra problem, more details
    4246. +
    4247. Generalizing the fitting procedure as a linear algebra problem
    4248. +
    4249. Generalizing the fitting procedure as a linear algebra problem
    4250. +
    4251. Optimizing our parameters
    4252. +
    4253. Our model for the nuclear binding energies
    4254. +
    4255. Optimizing our parameters, more details
    4256. +
    4257. Interpretations and optimizing our parameters
    4258. +
    4259. Interpretations and optimizing our parameters
    4260. +
    4261. Some useful matrix and vector expressions
    4262. +
    4263. Interpretations and optimizing our parameters
    4264. +
    4265. Own code for Ordinary Least Squares
    4266. +
    4267. Adding error analysis and training set up
    4268. +
    4269. The \( \chi^2 \) function
    4270. +
    4271. The \( \chi^2 \) function
    4272. +
    4273. The \( \chi^2 \) function
    4274. +
    4275. The \( \chi^2 \) function
    4276. +
    4277. The \( \chi^2 \) function
    4278. +
    4279. The \( \chi^2 \) function
    4280. +
    4281. Fitting an Equation of State for Dense Nuclear Matter
    4282. +
    4283. The code
    4284. +
    4285. Splitting our Data in Training and Test data
    4286. +
    4287. Exercises
    4288. +
    4289. Exercise 1: Setting up various Python environments
    4290. +
    4291. Exercise 2: making your own data and exploring scikit-learn
    4292. +
    4293. Exercise 3: Split data in test and training data
    4294. @@ -407,7 +391,7 @@ no setup and runs entirely in the cloud. Try it out!
    4295. 34
    4296. 35
    4297. ...
    4298. -
    4299. 66
    4300. +
    4301. 62
    4302. »
    4303. diff --git a/doc/pub/week34/html/._week34-bs026.html b/doc/pub/week34/html/._week34-bs026.html index af49df0fe..d1ed46eeb 100644 --- a/doc/pub/week34/html/._week34-bs026.html +++ b/doc/pub/week34/html/._week34-bs026.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    4304. Installing R, C++, cython or Julia
    4305. Installing R, C++, cython, Numba etc
    4306. Numpy examples and Important Matrix and vector handling packages
    4307. -
    4308. Basic Matrix Features
    4309. -
    4310.    Some famous Matrices
    4311. -
    4312.    More Basic Matrix Features
    4313. -
    4314. Numpy and arrays
    4315. -
    4316. Matrices in Python
    4317. -
    4318. Meet the Pandas
    4319. -
    4320. Friday August 27
    4321. -
    4322.    Simple linear regression model using scikit-learn
    4323. -
    4324.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4325. -
    4326.    Organizing our data
    4327. -
    4328.    Seeing the wood for the trees
    4329. -
    4330.    And what about using neural networks?
    4331. -
    4332. A first summary
    4333. -
    4334. Why Linear Regression (aka Ordinary Least Squares and family)
    4335. -
    4336. Regression analysis, overarching aims
    4337. -
    4338. Regression analysis, overarching aims II
    4339. -
    4340. Examples
    4341. -
    4342. General linear models
    4343. -
    4344. Rewriting the fitting procedure as a linear algebra problem
    4345. -
    4346. Rewriting the fitting procedure as a linear algebra problem, more details
    4347. -
    4348. Generalizing the fitting procedure as a linear algebra problem
    4349. -
    4350. Generalizing the fitting procedure as a linear algebra problem
    4351. -
    4352. Optimizing our parameters
    4353. -
    4354. Our model for the nuclear binding energies
    4355. -
    4356. Optimizing our parameters, more details
    4357. -
    4358. Interpretations and optimizing our parameters
    4359. -
    4360. Interpretations and optimizing our parameters
    4361. -
    4362. Some useful matrix and vector expressions
    4363. -
    4364. Interpretations and optimizing our parameters
    4365. -
    4366. Own code for Ordinary Least Squares
    4367. -
    4368. Adding error analysis and training set up
    4369. -
    4370. The \( \chi^2 \) function
    4371. -
    4372. The \( \chi^2 \) function
    4373. -
    4374. The \( \chi^2 \) function
    4375. -
    4376. The \( \chi^2 \) function
    4377. -
    4378. The \( \chi^2 \) function
    4379. -
    4380. The \( \chi^2 \) function
    4381. -
    4382. Fitting an Equation of State for Dense Nuclear Matter
    4383. -
    4384. The code
    4385. -
    4386. Splitting our Data in Training and Test data
    4387. -
    4388. Exercises
    4389. -
    4390. Exercise 1: Setting up various Python environments
    4391. -
    4392. Exercise 2: making your own data and exploring scikit-learn
    4393. -
    4394. Exercise 3: Normalizing our data
    4395. +
    4396. Numpy and arrays
    4397. +
    4398. Matrices in Python
    4399. +
    4400. Meet the Pandas
    4401. +
    4402.    Simple linear regression model using scikit-learn
    4403. +
    4404.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4405. +
    4406.    Organizing our data
    4407. +
    4408.    And what about using neural networks?
    4409. +
    4410. A first summary
    4411. +
    4412. Why Linear Regression (aka Ordinary Least Squares and family)
    4413. +
    4414. Regression analysis, overarching aims
    4415. +
    4416. Regression analysis, overarching aims II
    4417. +
    4418. Examples
    4419. +
    4420. General linear models
    4421. +
    4422. Rewriting the fitting procedure as a linear algebra problem
    4423. +
    4424. Rewriting the fitting procedure as a linear algebra problem, more details
    4425. +
    4426. Generalizing the fitting procedure as a linear algebra problem
    4427. +
    4428. Generalizing the fitting procedure as a linear algebra problem
    4429. +
    4430. Optimizing our parameters
    4431. +
    4432. Our model for the nuclear binding energies
    4433. +
    4434. Optimizing our parameters, more details
    4435. +
    4436. Interpretations and optimizing our parameters
    4437. +
    4438. Interpretations and optimizing our parameters
    4439. +
    4440. Some useful matrix and vector expressions
    4441. +
    4442. Interpretations and optimizing our parameters
    4443. +
    4444. Own code for Ordinary Least Squares
    4445. +
    4446. Adding error analysis and training set up
    4447. +
    4448. The \( \chi^2 \) function
    4449. +
    4450. The \( \chi^2 \) function
    4451. +
    4452. The \( \chi^2 \) function
    4453. +
    4454. The \( \chi^2 \) function
    4455. +
    4456. The \( \chi^2 \) function
    4457. +
    4458. The \( \chi^2 \) function
    4459. +
    4460. Fitting an Equation of State for Dense Nuclear Matter
    4461. +
    4462. The code
    4463. +
    4464. Splitting our Data in Training and Test data
    4465. +
    4466. Exercises
    4467. +
    4468. Exercise 1: Setting up various Python environments
    4469. +
    4470. Exercise 2: making your own data and exploring scikit-learn
    4471. +
    4472. Exercise 3: Split data in test and training data
    4473. @@ -393,7 +377,7 @@ MathJax.Hub.Config({
    4474. 35
    4475. 36
    4476. ...
    4477. -
    4478. 66
    4479. +
    4480. 62
    4481. »
    4482. diff --git a/doc/pub/week34/html/._week34-bs027.html b/doc/pub/week34/html/._week34-bs027.html index c91007472..d1aff13da 100644 --- a/doc/pub/week34/html/._week34-bs027.html +++ b/doc/pub/week34/html/._week34-bs027.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    4483. Installing R, C++, cython or Julia
    4484. Installing R, C++, cython, Numba etc
    4485. Numpy examples and Important Matrix and vector handling packages
    4486. -
    4487. Basic Matrix Features
    4488. -
    4489.    Some famous Matrices
    4490. -
    4491.    More Basic Matrix Features
    4492. -
    4493. Numpy and arrays
    4494. -
    4495. Matrices in Python
    4496. -
    4497. Meet the Pandas
    4498. -
    4499. Friday August 27
    4500. -
    4501.    Simple linear regression model using scikit-learn
    4502. -
    4503.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4504. -
    4505.    Organizing our data
    4506. -
    4507.    Seeing the wood for the trees
    4508. -
    4509.    And what about using neural networks?
    4510. -
    4511. A first summary
    4512. -
    4513. Why Linear Regression (aka Ordinary Least Squares and family)
    4514. -
    4515. Regression analysis, overarching aims
    4516. -
    4517. Regression analysis, overarching aims II
    4518. -
    4519. Examples
    4520. -
    4521. General linear models
    4522. -
    4523. Rewriting the fitting procedure as a linear algebra problem
    4524. -
    4525. Rewriting the fitting procedure as a linear algebra problem, more details
    4526. -
    4527. Generalizing the fitting procedure as a linear algebra problem
    4528. -
    4529. Generalizing the fitting procedure as a linear algebra problem
    4530. -
    4531. Optimizing our parameters
    4532. -
    4533. Our model for the nuclear binding energies
    4534. -
    4535. Optimizing our parameters, more details
    4536. -
    4537. Interpretations and optimizing our parameters
    4538. -
    4539. Interpretations and optimizing our parameters
    4540. -
    4541. Some useful matrix and vector expressions
    4542. -
    4543. Interpretations and optimizing our parameters
    4544. -
    4545. Own code for Ordinary Least Squares
    4546. -
    4547. Adding error analysis and training set up
    4548. -
    4549. The \( \chi^2 \) function
    4550. -
    4551. The \( \chi^2 \) function
    4552. -
    4553. The \( \chi^2 \) function
    4554. -
    4555. The \( \chi^2 \) function
    4556. -
    4557. The \( \chi^2 \) function
    4558. -
    4559. The \( \chi^2 \) function
    4560. -
    4561. Fitting an Equation of State for Dense Nuclear Matter
    4562. -
    4563. The code
    4564. -
    4565. Splitting our Data in Training and Test data
    4566. -
    4567. Exercises
    4568. -
    4569. Exercise 1: Setting up various Python environments
    4570. -
    4571. Exercise 2: making your own data and exploring scikit-learn
    4572. -
    4573. Exercise 3: Normalizing our data
    4574. +
    4575. Numpy and arrays
    4576. +
    4577. Matrices in Python
    4578. +
    4579. Meet the Pandas
    4580. +
    4581.    Simple linear regression model using scikit-learn
    4582. +
    4583.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4584. +
    4585.    Organizing our data
    4586. +
    4587.    And what about using neural networks?
    4588. +
    4589. A first summary
    4590. +
    4591. Why Linear Regression (aka Ordinary Least Squares and family)
    4592. +
    4593. Regression analysis, overarching aims
    4594. +
    4595. Regression analysis, overarching aims II
    4596. +
    4597. Examples
    4598. +
    4599. General linear models
    4600. +
    4601. Rewriting the fitting procedure as a linear algebra problem
    4602. +
    4603. Rewriting the fitting procedure as a linear algebra problem, more details
    4604. +
    4605. Generalizing the fitting procedure as a linear algebra problem
    4606. +
    4607. Generalizing the fitting procedure as a linear algebra problem
    4608. +
    4609. Optimizing our parameters
    4610. +
    4611. Our model for the nuclear binding energies
    4612. +
    4613. Optimizing our parameters, more details
    4614. +
    4615. Interpretations and optimizing our parameters
    4616. +
    4617. Interpretations and optimizing our parameters
    4618. +
    4619. Some useful matrix and vector expressions
    4620. +
    4621. Interpretations and optimizing our parameters
    4622. +
    4623. Own code for Ordinary Least Squares
    4624. +
    4625. Adding error analysis and training set up
    4626. +
    4627. The \( \chi^2 \) function
    4628. +
    4629. The \( \chi^2 \) function
    4630. +
    4631. The \( \chi^2 \) function
    4632. +
    4633. The \( \chi^2 \) function
    4634. +
    4635. The \( \chi^2 \) function
    4636. +
    4637. The \( \chi^2 \) function
    4638. +
    4639. Fitting an Equation of State for Dense Nuclear Matter
    4640. +
    4641. The code
    4642. +
    4643. Splitting our Data in Training and Test data
    4644. +
    4645. Exercises
    4646. +
    4647. Exercise 1: Setting up various Python environments
    4648. +
    4649. Exercise 2: making your own data and exploring scikit-learn
    4650. +
    4651. Exercise 3: Split data in test and training data
    4652. @@ -358,8 +342,8 @@ use Python during our lectures and in various projects and exercises. Those of you already familiar with R should feel free to continue using R, keeping however an eye on the parallel Python set ups. Similarly, if you are a -Python afecionado, feel free to explore R as well. Jupyter/Ipython -notebook allows you to run R codes interactively in your +Python afecionado, feel free to explore R as well. Jupyter(Julia, Python and R) /Ipython +notebook allows you to run R codes and Julia codes interactively in your browser. The software library R is really tailored for statistical data analysis and allows for an easy usage of the tools and algorithms we will discuss in these lectures. @@ -394,7 +378,7 @@ lectures.
    4653. 36
    4654. 37
    4655. ...
    4656. -
    4657. 66
    4658. +
    4659. 62
    4660. »
    4661. diff --git a/doc/pub/week34/html/._week34-bs028.html b/doc/pub/week34/html/._week34-bs028.html index 7eb6e2136..5d204ba8a 100644 --- a/doc/pub/week34/html/._week34-bs028.html +++ b/doc/pub/week34/html/._week34-bs028.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    4662. Installing R, C++, cython or Julia
    4663. Installing R, C++, cython, Numba etc
    4664. Numpy examples and Important Matrix and vector handling packages
    4665. -
    4666. Basic Matrix Features
    4667. -
    4668.    Some famous Matrices
    4669. -
    4670.    More Basic Matrix Features
    4671. -
    4672. Numpy and arrays
    4673. -
    4674. Matrices in Python
    4675. -
    4676. Meet the Pandas
    4677. -
    4678. Friday August 27
    4679. -
    4680.    Simple linear regression model using scikit-learn
    4681. -
    4682.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4683. -
    4684.    Organizing our data
    4685. -
    4686.    Seeing the wood for the trees
    4687. -
    4688.    And what about using neural networks?
    4689. -
    4690. A first summary
    4691. -
    4692. Why Linear Regression (aka Ordinary Least Squares and family)
    4693. -
    4694. Regression analysis, overarching aims
    4695. -
    4696. Regression analysis, overarching aims II
    4697. -
    4698. Examples
    4699. -
    4700. General linear models
    4701. -
    4702. Rewriting the fitting procedure as a linear algebra problem
    4703. -
    4704. Rewriting the fitting procedure as a linear algebra problem, more details
    4705. -
    4706. Generalizing the fitting procedure as a linear algebra problem
    4707. -
    4708. Generalizing the fitting procedure as a linear algebra problem
    4709. -
    4710. Optimizing our parameters
    4711. -
    4712. Our model for the nuclear binding energies
    4713. -
    4714. Optimizing our parameters, more details
    4715. -
    4716. Interpretations and optimizing our parameters
    4717. -
    4718. Interpretations and optimizing our parameters
    4719. -
    4720. Some useful matrix and vector expressions
    4721. -
    4722. Interpretations and optimizing our parameters
    4723. -
    4724. Own code for Ordinary Least Squares
    4725. -
    4726. Adding error analysis and training set up
    4727. -
    4728. The \( \chi^2 \) function
    4729. -
    4730. The \( \chi^2 \) function
    4731. -
    4732. The \( \chi^2 \) function
    4733. -
    4734. The \( \chi^2 \) function
    4735. -
    4736. The \( \chi^2 \) function
    4737. -
    4738. The \( \chi^2 \) function
    4739. -
    4740. Fitting an Equation of State for Dense Nuclear Matter
    4741. -
    4742. The code
    4743. -
    4744. Splitting our Data in Training and Test data
    4745. -
    4746. Exercises
    4747. -
    4748. Exercise 1: Setting up various Python environments
    4749. -
    4750. Exercise 2: making your own data and exploring scikit-learn
    4751. -
    4752. Exercise 3: Normalizing our data
    4753. +
    4754. Numpy and arrays
    4755. +
    4756. Matrices in Python
    4757. +
    4758. Meet the Pandas
    4759. +
    4760.    Simple linear regression model using scikit-learn
    4761. +
    4762.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4763. +
    4764.    Organizing our data
    4765. +
    4766.    And what about using neural networks?
    4767. +
    4768. A first summary
    4769. +
    4770. Why Linear Regression (aka Ordinary Least Squares and family)
    4771. +
    4772. Regression analysis, overarching aims
    4773. +
    4774. Regression analysis, overarching aims II
    4775. +
    4776. Examples
    4777. +
    4778. General linear models
    4779. +
    4780. Rewriting the fitting procedure as a linear algebra problem
    4781. +
    4782. Rewriting the fitting procedure as a linear algebra problem, more details
    4783. +
    4784. Generalizing the fitting procedure as a linear algebra problem
    4785. +
    4786. Generalizing the fitting procedure as a linear algebra problem
    4787. +
    4788. Optimizing our parameters
    4789. +
    4790. Our model for the nuclear binding energies
    4791. +
    4792. Optimizing our parameters, more details
    4793. +
    4794. Interpretations and optimizing our parameters
    4795. +
    4796. Interpretations and optimizing our parameters
    4797. +
    4798. Some useful matrix and vector expressions
    4799. +
    4800. Interpretations and optimizing our parameters
    4801. +
    4802. Own code for Ordinary Least Squares
    4803. +
    4804. Adding error analysis and training set up
    4805. +
    4806. The \( \chi^2 \) function
    4807. +
    4808. The \( \chi^2 \) function
    4809. +
    4810. The \( \chi^2 \) function
    4811. +
    4812. The \( \chi^2 \) function
    4813. +
    4814. The \( \chi^2 \) function
    4815. +
    4816. The \( \chi^2 \) function
    4817. +
    4818. Fitting an Equation of State for Dense Nuclear Matter
    4819. +
    4820. The code
    4821. +
    4822. Splitting our Data in Training and Test data
    4823. +
    4824. Exercises
    4825. +
    4826. Exercise 1: Setting up various Python environments
    4827. +
    4828. Exercise 2: making your own data and exploring scikit-learn
    4829. +
    4830. Exercise 3: Split data in test and training data
    4831. @@ -397,10 +381,7 @@ further processing. For example, convert to latex as

      And to add more versatility, the Python package SymPy is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python.

      -

      Finally, if you wish to use the light mark-up language -doconce you can convert a standard ascii text file into various HTML -formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using doconce. -

      +

      Finally, we recommend strongly using Autograd or JAX for automatic differentiation.

      @@ -427,7 +408,7 @@ formats, ipython notebooks, latex files, pdf files etc with minimal edits. These

    4832. 37
    4833. 38
    4834. ...
    4835. -
    4836. 66
    4837. +
    4838. 62
    4839. »
    4840. diff --git a/doc/pub/week34/html/._week34-bs029.html b/doc/pub/week34/html/._week34-bs029.html index 167881953..375a1c0c6 100644 --- a/doc/pub/week34/html/._week34-bs029.html +++ b/doc/pub/week34/html/._week34-bs029.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    4841. Installing R, C++, cython or Julia
    4842. Installing R, C++, cython, Numba etc
    4843. Numpy examples and Important Matrix and vector handling packages
    4844. -
    4845. Basic Matrix Features
    4846. -
    4847.    Some famous Matrices
    4848. -
    4849.    More Basic Matrix Features
    4850. -
    4851. Numpy and arrays
    4852. -
    4853. Matrices in Python
    4854. -
    4855. Meet the Pandas
    4856. -
    4857. Friday August 27
    4858. -
    4859.    Simple linear regression model using scikit-learn
    4860. -
    4861.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4862. -
    4863.    Organizing our data
    4864. -
    4865.    Seeing the wood for the trees
    4866. -
    4867.    And what about using neural networks?
    4868. -
    4869. A first summary
    4870. -
    4871. Why Linear Regression (aka Ordinary Least Squares and family)
    4872. -
    4873. Regression analysis, overarching aims
    4874. -
    4875. Regression analysis, overarching aims II
    4876. -
    4877. Examples
    4878. -
    4879. General linear models
    4880. -
    4881. Rewriting the fitting procedure as a linear algebra problem
    4882. -
    4883. Rewriting the fitting procedure as a linear algebra problem, more details
    4884. -
    4885. Generalizing the fitting procedure as a linear algebra problem
    4886. -
    4887. Generalizing the fitting procedure as a linear algebra problem
    4888. -
    4889. Optimizing our parameters
    4890. -
    4891. Our model for the nuclear binding energies
    4892. -
    4893. Optimizing our parameters, more details
    4894. -
    4895. Interpretations and optimizing our parameters
    4896. -
    4897. Interpretations and optimizing our parameters
    4898. -
    4899. Some useful matrix and vector expressions
    4900. -
    4901. Interpretations and optimizing our parameters
    4902. -
    4903. Own code for Ordinary Least Squares
    4904. -
    4905. Adding error analysis and training set up
    4906. -
    4907. The \( \chi^2 \) function
    4908. -
    4909. The \( \chi^2 \) function
    4910. -
    4911. The \( \chi^2 \) function
    4912. -
    4913. The \( \chi^2 \) function
    4914. -
    4915. The \( \chi^2 \) function
    4916. -
    4917. The \( \chi^2 \) function
    4918. -
    4919. Fitting an Equation of State for Dense Nuclear Matter
    4920. -
    4921. The code
    4922. -
    4923. Splitting our Data in Training and Test data
    4924. -
    4925. Exercises
    4926. -
    4927. Exercise 1: Setting up various Python environments
    4928. -
    4929. Exercise 2: making your own data and exploring scikit-learn
    4930. -
    4931. Exercise 3: Normalizing our data
    4932. +
    4933. Numpy and arrays
    4934. +
    4935. Matrices in Python
    4936. +
    4937. Meet the Pandas
    4938. +
    4939.    Simple linear regression model using scikit-learn
    4940. +
    4941.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    4942. +
    4943.    Organizing our data
    4944. +
    4945.    And what about using neural networks?
    4946. +
    4947. A first summary
    4948. +
    4949. Why Linear Regression (aka Ordinary Least Squares and family)
    4950. +
    4951. Regression analysis, overarching aims
    4952. +
    4953. Regression analysis, overarching aims II
    4954. +
    4955. Examples
    4956. +
    4957. General linear models
    4958. +
    4959. Rewriting the fitting procedure as a linear algebra problem
    4960. +
    4961. Rewriting the fitting procedure as a linear algebra problem, more details
    4962. +
    4963. Generalizing the fitting procedure as a linear algebra problem
    4964. +
    4965. Generalizing the fitting procedure as a linear algebra problem
    4966. +
    4967. Optimizing our parameters
    4968. +
    4969. Our model for the nuclear binding energies
    4970. +
    4971. Optimizing our parameters, more details
    4972. +
    4973. Interpretations and optimizing our parameters
    4974. +
    4975. Interpretations and optimizing our parameters
    4976. +
    4977. Some useful matrix and vector expressions
    4978. +
    4979. Interpretations and optimizing our parameters
    4980. +
    4981. Own code for Ordinary Least Squares
    4982. +
    4983. Adding error analysis and training set up
    4984. +
    4985. The \( \chi^2 \) function
    4986. +
    4987. The \( \chi^2 \) function
    4988. +
    4989. The \( \chi^2 \) function
    4990. +
    4991. The \( \chi^2 \) function
    4992. +
    4993. The \( \chi^2 \) function
    4994. +
    4995. The \( \chi^2 \) function
    4996. +
    4997. Fitting an Equation of State for Dense Nuclear Matter
    4998. +
    4999. The code
    5000. +
    5001. Splitting our Data in Training and Test data
    5002. +
    5003. Exercises
    5004. +
    5005. Exercise 1: Setting up various Python environments
    5006. +
    5007. Exercise 2: making your own data and exploring scikit-learn
    5008. +
    5009. Exercise 3: Split data in test and training data
    5010. @@ -389,7 +373,7 @@ developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly he
    5011. 38
    5012. 39
    5013. ...
    5014. -
    5015. 66
    5016. +
    5017. 62
    5018. »
    5019. diff --git a/doc/pub/week34/html/._week34-bs030.html b/doc/pub/week34/html/._week34-bs030.html index 72f73c715..6430d9ef8 100644 --- a/doc/pub/week34/html/._week34-bs030.html +++ b/doc/pub/week34/html/._week34-bs030.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    5020. Installing R, C++, cython or Julia
    5021. Installing R, C++, cython, Numba etc
    5022. Numpy examples and Important Matrix and vector handling packages
    5023. -
    5024. Basic Matrix Features
    5025. -
    5026.    Some famous Matrices
    5027. -
    5028.    More Basic Matrix Features
    5029. -
    5030. Numpy and arrays
    5031. -
    5032. Matrices in Python
    5033. -
    5034. Meet the Pandas
    5035. -
    5036. Friday August 27
    5037. -
    5038.    Simple linear regression model using scikit-learn
    5039. -
    5040.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5041. -
    5042.    Organizing our data
    5043. -
    5044.    Seeing the wood for the trees
    5045. -
    5046.    And what about using neural networks?
    5047. -
    5048. A first summary
    5049. -
    5050. Why Linear Regression (aka Ordinary Least Squares and family)
    5051. -
    5052. Regression analysis, overarching aims
    5053. -
    5054. Regression analysis, overarching aims II
    5055. -
    5056. Examples
    5057. -
    5058. General linear models
    5059. -
    5060. Rewriting the fitting procedure as a linear algebra problem
    5061. -
    5062. Rewriting the fitting procedure as a linear algebra problem, more details
    5063. -
    5064. Generalizing the fitting procedure as a linear algebra problem
    5065. -
    5066. Generalizing the fitting procedure as a linear algebra problem
    5067. -
    5068. Optimizing our parameters
    5069. -
    5070. Our model for the nuclear binding energies
    5071. -
    5072. Optimizing our parameters, more details
    5073. -
    5074. Interpretations and optimizing our parameters
    5075. -
    5076. Interpretations and optimizing our parameters
    5077. -
    5078. Some useful matrix and vector expressions
    5079. -
    5080. Interpretations and optimizing our parameters
    5081. -
    5082. Own code for Ordinary Least Squares
    5083. -
    5084. Adding error analysis and training set up
    5085. -
    5086. The \( \chi^2 \) function
    5087. -
    5088. The \( \chi^2 \) function
    5089. -
    5090. The \( \chi^2 \) function
    5091. -
    5092. The \( \chi^2 \) function
    5093. -
    5094. The \( \chi^2 \) function
    5095. -
    5096. The \( \chi^2 \) function
    5097. -
    5098. Fitting an Equation of State for Dense Nuclear Matter
    5099. -
    5100. The code
    5101. -
    5102. Splitting our Data in Training and Test data
    5103. -
    5104. Exercises
    5105. -
    5106. Exercise 1: Setting up various Python environments
    5107. -
    5108. Exercise 2: making your own data and exploring scikit-learn
    5109. -
    5110. Exercise 3: Normalizing our data
    5111. +
    5112. Numpy and arrays
    5113. +
    5114. Matrices in Python
    5115. +
    5116. Meet the Pandas
    5117. +
    5118.    Simple linear regression model using scikit-learn
    5119. +
    5120.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5121. +
    5122.    Organizing our data
    5123. +
    5124.    And what about using neural networks?
    5125. +
    5126. A first summary
    5127. +
    5128. Why Linear Regression (aka Ordinary Least Squares and family)
    5129. +
    5130. Regression analysis, overarching aims
    5131. +
    5132. Regression analysis, overarching aims II
    5133. +
    5134. Examples
    5135. +
    5136. General linear models
    5137. +
    5138. Rewriting the fitting procedure as a linear algebra problem
    5139. +
    5140. Rewriting the fitting procedure as a linear algebra problem, more details
    5141. +
    5142. Generalizing the fitting procedure as a linear algebra problem
    5143. +
    5144. Generalizing the fitting procedure as a linear algebra problem
    5145. +
    5146. Optimizing our parameters
    5147. +
    5148. Our model for the nuclear binding energies
    5149. +
    5150. Optimizing our parameters, more details
    5151. +
    5152. Interpretations and optimizing our parameters
    5153. +
    5154. Interpretations and optimizing our parameters
    5155. +
    5156. Some useful matrix and vector expressions
    5157. +
    5158. Interpretations and optimizing our parameters
    5159. +
    5160. Own code for Ordinary Least Squares
    5161. +
    5162. Adding error analysis and training set up
    5163. +
    5164. The \( \chi^2 \) function
    5165. +
    5166. The \( \chi^2 \) function
    5167. +
    5168. The \( \chi^2 \) function
    5169. +
    5170. The \( \chi^2 \) function
    5171. +
    5172. The \( \chi^2 \) function
    5173. +
    5174. The \( \chi^2 \) function
    5175. +
    5176. Fitting an Equation of State for Dense Nuclear Matter
    5177. +
    5178. The code
    5179. +
    5180. Splitting our Data in Training and Test data
    5181. +
    5182. Exercises
    5183. +
    5184. Exercise 1: Setting up various Python environments
    5185. +
    5186. Exercise 2: making your own data and exploring scikit-learn
    5187. +
    5188. Exercise 3: Split data in test and training data
    5189. @@ -351,50 +335,229 @@ MathJax.Hub.Config({

       

       

       

      -

      Basic Matrix Features

      - -
      -
      - -$$ - \mathbf{A} = - \begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\ - a_{21} & a_{22} & a_{23} & a_{24} \\ - a_{31} & a_{32} & a_{33} & a_{34} \\ - a_{41} & a_{42} & a_{43} & a_{44} - \end{bmatrix}\qquad -\mathbf{I} = - \begin{bmatrix} 1 & 0 & 0 & 0 \\ - 0 & 1 & 0 & 0 \\ - 0 & 0 & 1 & 0 \\ - 0 & 0 & 0 & 1 - \end{bmatrix} -$$ - -

      The inverse of a matrix is defined by

      - -$$ -\mathbf{A}^{-1} \cdot \mathbf{A} = I -$$ +

      Numpy and arrays

      +

      Numpy provides an easy way to handle arrays in Python. The standard way to import this library is as

      -
      -
      - - - - - - - - - - - -
      Relations Name matrix elements
      \( A=A^{T} \) symmetric \( a_{ij}=a_{ji} \)
      \( A=\left (A^{T}\right )^{-1} \) real orthogonal \( \sum_k a_{ik}a_{jk}=\sum_k a_{ki} a_{kj}=\delta_{ij} \)
      \( A=A^* \) real matrix \( a_{ij}=a_{ij}^* \)
      \( A=A^{\dagger} \) hermitian \( a_{ij}=a_{ji}^* \)
      \( A=\left(A^{\dagger}\right )^{-1} \) unitary \( \sum_k a_{ik}a_{jk}^*=\sum_k a_{ki}^* a_{kj}=\delta_{ij} \)
      -
      -
      + +
      +
      +
      +
      +
      +
      import numpy as np
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Here follows a simple example where we set up an array of ten elements, all determined by random numbers drawn according to the normal distribution,

      + + +
      +
      +
      +
      +
      +
      n = 10
      +x = np.random.normal(size=n)
      +print(x)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      We defined a vector \( x \) with \( n=10 \) elements with its values given by the Normal distribution \( N(0,1) \). +Another alternative is to declare a vector as follows +

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +x = np.array([1, 2, 3])
      +print(x)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Here we have defined a vector with three elements, with \( x_0=1 \), \( x_1=2 \) and \( x_2=3 \). Note that both Python and C++ +start numbering array elements from \( 0 \) and on. This means that a vector with \( n \) elements has a sequence of entities \( x_0, x_1, x_2, \dots, x_{n-1} \). We could also let (recommended) Numpy to compute the logarithms of a specific array as +

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +x = np.log(np.array([4, 7, 8]))
      +print(x)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      In the last example we used Numpy's unary function \( np.log \). This function is +highly tuned to compute array elements since the code is vectorized +and does not require looping. We normaly recommend that you use the +Numpy intrinsic functions instead of the corresponding log function +from Python's math module. The looping is done explicitely by the +np.log function. The alternative, and slower way to compute the +logarithms of a vector would be to write +

      + + + +
      +
      +
      +
      +
      +
      import numpy as np
      +from math import log
      +x = np.array([4, 7, 8])
      +for i in range(0, len(x)):
      +    x[i] = log(x[i])
      +print(x)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      We note that our code is much longer already and we need to import the log function from the math module. +The attentive reader will also notice that the output is \( [1, 1, 2] \). Python interprets automagically our numbers as integers (like the automatic keyword in C++). To change this we could define our array elements to be double precision numbers as +

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +x = np.log(np.array([4, 7, 8], dtype = np.float64))
      +print(x)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      or simply write them as double precision numbers (Python uses 64 bits as default for floating point type variables), that is

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +x = np.log(np.array([4.0, 7.0, 8.0]))
      +print(x)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      To check the number of bytes (remember that one byte contains eight bits for double precision variables), you can use simple use the itemsize functionality (the array \( x \) is actually an object which inherits the functionalities defined in Numpy) as

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +x = np.log(np.array([4.0, 7.0, 8.0]))
      +print(x.itemsize)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      @@ -423,7 +586,7 @@ $$
    5190. 39
    5191. 40
    5192. ...
    5193. -
    5194. 66
    5195. +
    5196. 62
    5197. »
    5198. diff --git a/doc/pub/week34/html/._week34-bs031.html b/doc/pub/week34/html/._week34-bs031.html index 988bbf56d..84bd312b1 100644 --- a/doc/pub/week34/html/._week34-bs031.html +++ b/doc/pub/week34/html/._week34-bs031.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    5199. Installing R, C++, cython or Julia
    5200. Installing R, C++, cython, Numba etc
    5201. Numpy examples and Important Matrix and vector handling packages
    5202. -
    5203. Basic Matrix Features
    5204. -
    5205.    Some famous Matrices
    5206. -
    5207.    More Basic Matrix Features
    5208. -
    5209. Numpy and arrays
    5210. -
    5211. Matrices in Python
    5212. -
    5213. Meet the Pandas
    5214. -
    5215. Friday August 27
    5216. -
    5217.    Simple linear regression model using scikit-learn
    5218. -
    5219.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5220. -
    5221.    Organizing our data
    5222. -
    5223.    Seeing the wood for the trees
    5224. -
    5225.    And what about using neural networks?
    5226. -
    5227. A first summary
    5228. -
    5229. Why Linear Regression (aka Ordinary Least Squares and family)
    5230. -
    5231. Regression analysis, overarching aims
    5232. -
    5233. Regression analysis, overarching aims II
    5234. -
    5235. Examples
    5236. -
    5237. General linear models
    5238. -
    5239. Rewriting the fitting procedure as a linear algebra problem
    5240. -
    5241. Rewriting the fitting procedure as a linear algebra problem, more details
    5242. -
    5243. Generalizing the fitting procedure as a linear algebra problem
    5244. -
    5245. Generalizing the fitting procedure as a linear algebra problem
    5246. -
    5247. Optimizing our parameters
    5248. -
    5249. Our model for the nuclear binding energies
    5250. -
    5251. Optimizing our parameters, more details
    5252. -
    5253. Interpretations and optimizing our parameters
    5254. -
    5255. Interpretations and optimizing our parameters
    5256. -
    5257. Some useful matrix and vector expressions
    5258. -
    5259. Interpretations and optimizing our parameters
    5260. -
    5261. Own code for Ordinary Least Squares
    5262. -
    5263. Adding error analysis and training set up
    5264. -
    5265. The \( \chi^2 \) function
    5266. -
    5267. The \( \chi^2 \) function
    5268. -
    5269. The \( \chi^2 \) function
    5270. -
    5271. The \( \chi^2 \) function
    5272. -
    5273. The \( \chi^2 \) function
    5274. -
    5275. The \( \chi^2 \) function
    5276. -
    5277. Fitting an Equation of State for Dense Nuclear Matter
    5278. -
    5279. The code
    5280. -
    5281. Splitting our Data in Training and Test data
    5282. -
    5283. Exercises
    5284. -
    5285. Exercise 1: Setting up various Python environments
    5286. -
    5287. Exercise 2: making your own data and exploring scikit-learn
    5288. -
    5289. Exercise 3: Normalizing our data
    5290. +
    5291. Numpy and arrays
    5292. +
    5293. Matrices in Python
    5294. +
    5295. Meet the Pandas
    5296. +
    5297.    Simple linear regression model using scikit-learn
    5298. +
    5299.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5300. +
    5301.    Organizing our data
    5302. +
    5303.    And what about using neural networks?
    5304. +
    5305. A first summary
    5306. +
    5307. Why Linear Regression (aka Ordinary Least Squares and family)
    5308. +
    5309. Regression analysis, overarching aims
    5310. +
    5311. Regression analysis, overarching aims II
    5312. +
    5313. Examples
    5314. +
    5315. General linear models
    5316. +
    5317. Rewriting the fitting procedure as a linear algebra problem
    5318. +
    5319. Rewriting the fitting procedure as a linear algebra problem, more details
    5320. +
    5321. Generalizing the fitting procedure as a linear algebra problem
    5322. +
    5323. Generalizing the fitting procedure as a linear algebra problem
    5324. +
    5325. Optimizing our parameters
    5326. +
    5327. Our model for the nuclear binding energies
    5328. +
    5329. Optimizing our parameters, more details
    5330. +
    5331. Interpretations and optimizing our parameters
    5332. +
    5333. Interpretations and optimizing our parameters
    5334. +
    5335. Some useful matrix and vector expressions
    5336. +
    5337. Interpretations and optimizing our parameters
    5338. +
    5339. Own code for Ordinary Least Squares
    5340. +
    5341. Adding error analysis and training set up
    5342. +
    5343. The \( \chi^2 \) function
    5344. +
    5345. The \( \chi^2 \) function
    5346. +
    5347. The \( \chi^2 \) function
    5348. +
    5349. The \( \chi^2 \) function
    5350. +
    5351. The \( \chi^2 \) function
    5352. +
    5353. The \( \chi^2 \) function
    5354. +
    5355. Fitting an Equation of State for Dense Nuclear Matter
    5356. +
    5357. The code
    5358. +
    5359. Splitting our Data in Training and Test data
    5360. +
    5361. Exercises
    5362. +
    5363. Exercise 1: Setting up various Python environments
    5364. +
    5365. Exercise 2: making your own data and exploring scikit-learn
    5366. +
    5367. Exercise 3: Split data in test and training data
    5368. @@ -351,19 +335,277 @@ MathJax.Hub.Config({

       

       

       

      -

      Some famous Matrices

      +

      Matrices in Python

      + +

      Having defined vectors, we are now ready to try out matrices. We can +define a \( 3 \times 3 \) real matrix \( \boldsymbol{A} \) as (recall that we user +lowercase letters for vectors and uppercase letters for matrices) +

      + + + +
      +
      +
      +
      +
      +
      import numpy as np
      +A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
      +print(A)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      If we use the shape function we would get \( (3, 3) \) as output, that is verifying that our matrix is a \( 3\times 3 \) matrix. We can slice the matrix and print for example the first column (Python organized matrix elements in a row-major order, see below) as

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
      +# print the first column, row-major order and elements start with 0
      +print(A[:,0]) 
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      We can continue this was by printing out other columns or rows. The example here prints out the second column

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
      +# print the first column, row-major order and elements start with 0
      +print(A[1,:]) 
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Numpy contains many other functionalities that allow us to slice, subdivide etc etc arrays. We strongly recommend that you look up the Numpy website for more details. Useful functions when defining a matrix are the np.zeros function which declares a matrix of a given dimension and sets all elements to zero

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +n = 10
      +# define a matrix of dimension 10 x 10 and set all elements to zero
      +A = np.zeros( (n, n) )
      +print(A) 
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      or initializing all elements to

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +n = 10
      +# define a matrix of dimension 10 x 10 and set all elements to one
      +A = np.ones( (n, n) )
      +print(A) 
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      or as unitarily distributed random numbers (see the material on random number generators in the statistics part)

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +n = 10
      +# define a matrix of dimension 10 x 10 and set all elements to random numbers with x \in [0, 1]
      +A = np.random.rand(n, n)
      +print(A) 
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      As we will see throughout these lectures, there are several extremely useful functionalities in Numpy. +As an example, consider the discussion of the covariance matrix. Suppose we have defined three vectors +\( \boldsymbol{x}, \boldsymbol{y}, \boldsymbol{z} \) with \( n \) elements each. The covariance matrix is defined as +

      +$$ +\boldsymbol{\Sigma} = \begin{bmatrix} \sigma_{xx} & \sigma_{xy} & \sigma_{xz} \\ + \sigma_{yx} & \sigma_{yy} & \sigma_{yz} \\ + \sigma_{zx} & \sigma_{zy} & \sigma_{zz} + \end{bmatrix}, +$$ + +

      where for example

      +$$ +\sigma_{xy} =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ + +

      The Numpy function np.cov calculates the covariance elements using the factor \( 1/(n-1) \) instead of \( 1/n \) since it assumes we do not have the exact mean values. +The following simple function uses the np.vstack function which takes each vector of dimension \( 1\times n \) and produces a \( 3\times n \) matrix \( \boldsymbol{W} \) +

      +$$ +\boldsymbol{W} = \begin{bmatrix} x_0 & x_1 & x_2 & \dots & x_{n-2} & x_{n-1} \\ + y_0 & y_1 & y_2 & \dots & y_{n-2} & y_{n-1} \\ + z_0 & z_1 & z_2 & \dots & z_{n-2} & z_{n-1} \\ + \end{bmatrix}, +$$ + +

      which in turn is converted into into the \( 3\times 3 \) covariance matrix +\( \boldsymbol{\Sigma} \) via the Numpy function np.cov(). We note that we can also calculate +the mean value of each set of samples \( \boldsymbol{x} \) etc using the Numpy +function np.mean(x). We can also extract the eigenvalues of the +covariance matrix through the np.linalg.eig() function. +

      + + + +
      +
      +
      +
      +
      +
      # Importing various packages
      +import numpy as np
      +
      +n = 100
      +x = np.random.normal(size=n)
      +print(np.mean(x))
      +y = 4+3*x+np.random.normal(size=n)
      +print(np.mean(y))
      +z = x**3+np.random.normal(size=n)
      +print(np.mean(z))
      +W = np.vstack((x, y, z))
      +Sigma = np.cov(W)
      +print(Sigma)
      +Eigvals, Eigvecs = np.linalg.eig(Sigma)
      +print(Eigvals)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +
      +
      +
      +
      +
      +
      import numpy as np
      +import matplotlib.pyplot as plt
      +from scipy import sparse
      +eye = np.eye(4)
      +print(eye)
      +sparse_mtx = sparse.csr_matrix(eye)
      +print(sparse_mtx)
      +x = np.linspace(-10,10,100)
      +y = np.sin(x)
      +plt.plot(x,y,marker='x')
      +plt.show()
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + -
        -
      • Diagonal if \( a_{ij}=0 \) for \( i\ne j \)
      • -
      • Upper triangular if \( a_{ij}=0 \) for \( i>j \)
      • -
      • Lower triangular if \( a_{ij}=0 \) for \( i < j \)
      • -
      • Upper Hessenberg if \( a_{ij}=0 \) for \( i>j+1 \)
      • -
      • Lower Hessenberg if \( a_{ij}=0 \) for \( i < j+1 \)
      • -
      • Tridiagonal if \( a_{ij}=0 \) for \( |i -j|>1 \)
      • -
      • Lower banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i>j+p \)
      • -
      • Upper banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i < j+p \)
      • -
      • Banded, block upper triangular, block lower triangular....
      • -

        @@ -389,7 +631,7 @@ MathJax.Hub.Config({
      • 40
      • 41
      • ...
      • -
      • 66
      • +
      • 62
      • »
      diff --git a/doc/pub/week34/html/._week34-bs032.html b/doc/pub/week34/html/._week34-bs032.html index f53225b17..cfaccfe40 100644 --- a/doc/pub/week34/html/._week34-bs032.html +++ b/doc/pub/week34/html/._week34-bs032.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    5369. Installing R, C++, cython or Julia
    5370. Installing R, C++, cython, Numba etc
    5371. Numpy examples and Important Matrix and vector handling packages
    5372. -
    5373. Basic Matrix Features
    5374. -
    5375.    Some famous Matrices
    5376. -
    5377.    More Basic Matrix Features
    5378. -
    5379. Numpy and arrays
    5380. -
    5381. Matrices in Python
    5382. -
    5383. Meet the Pandas
    5384. -
    5385. Friday August 27
    5386. -
    5387.    Simple linear regression model using scikit-learn
    5388. -
    5389.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5390. -
    5391.    Organizing our data
    5392. -
    5393.    Seeing the wood for the trees
    5394. -
    5395.    And what about using neural networks?
    5396. -
    5397. A first summary
    5398. -
    5399. Why Linear Regression (aka Ordinary Least Squares and family)
    5400. -
    5401. Regression analysis, overarching aims
    5402. -
    5403. Regression analysis, overarching aims II
    5404. -
    5405. Examples
    5406. -
    5407. General linear models
    5408. -
    5409. Rewriting the fitting procedure as a linear algebra problem
    5410. -
    5411. Rewriting the fitting procedure as a linear algebra problem, more details
    5412. -
    5413. Generalizing the fitting procedure as a linear algebra problem
    5414. -
    5415. Generalizing the fitting procedure as a linear algebra problem
    5416. -
    5417. Optimizing our parameters
    5418. -
    5419. Our model for the nuclear binding energies
    5420. -
    5421. Optimizing our parameters, more details
    5422. -
    5423. Interpretations and optimizing our parameters
    5424. -
    5425. Interpretations and optimizing our parameters
    5426. -
    5427. Some useful matrix and vector expressions
    5428. -
    5429. Interpretations and optimizing our parameters
    5430. -
    5431. Own code for Ordinary Least Squares
    5432. -
    5433. Adding error analysis and training set up
    5434. -
    5435. The \( \chi^2 \) function
    5436. -
    5437. The \( \chi^2 \) function
    5438. -
    5439. The \( \chi^2 \) function
    5440. -
    5441. The \( \chi^2 \) function
    5442. -
    5443. The \( \chi^2 \) function
    5444. -
    5445. The \( \chi^2 \) function
    5446. -
    5447. Fitting an Equation of State for Dense Nuclear Matter
    5448. -
    5449. The code
    5450. -
    5451. Splitting our Data in Training and Test data
    5452. -
    5453. Exercises
    5454. -
    5455. Exercise 1: Setting up various Python environments
    5456. -
    5457. Exercise 2: making your own data and exploring scikit-learn
    5458. -
    5459. Exercise 3: Normalizing our data
    5460. +
    5461. Numpy and arrays
    5462. +
    5463. Matrices in Python
    5464. +
    5465. Meet the Pandas
    5466. +
    5467.    Simple linear regression model using scikit-learn
    5468. +
    5469.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5470. +
    5471.    Organizing our data
    5472. +
    5473.    And what about using neural networks?
    5474. +
    5475. A first summary
    5476. +
    5477. Why Linear Regression (aka Ordinary Least Squares and family)
    5478. +
    5479. Regression analysis, overarching aims
    5480. +
    5481. Regression analysis, overarching aims II
    5482. +
    5483. Examples
    5484. +
    5485. General linear models
    5486. +
    5487. Rewriting the fitting procedure as a linear algebra problem
    5488. +
    5489. Rewriting the fitting procedure as a linear algebra problem, more details
    5490. +
    5491. Generalizing the fitting procedure as a linear algebra problem
    5492. +
    5493. Generalizing the fitting procedure as a linear algebra problem
    5494. +
    5495. Optimizing our parameters
    5496. +
    5497. Our model for the nuclear binding energies
    5498. +
    5499. Optimizing our parameters, more details
    5500. +
    5501. Interpretations and optimizing our parameters
    5502. +
    5503. Interpretations and optimizing our parameters
    5504. +
    5505. Some useful matrix and vector expressions
    5506. +
    5507. Interpretations and optimizing our parameters
    5508. +
    5509. Own code for Ordinary Least Squares
    5510. +
    5511. Adding error analysis and training set up
    5512. +
    5513. The \( \chi^2 \) function
    5514. +
    5515. The \( \chi^2 \) function
    5516. +
    5517. The \( \chi^2 \) function
    5518. +
    5519. The \( \chi^2 \) function
    5520. +
    5521. The \( \chi^2 \) function
    5522. +
    5523. The \( \chi^2 \) function
    5524. +
    5525. Fitting an Equation of State for Dense Nuclear Matter
    5526. +
    5527. The code
    5528. +
    5529. Splitting our Data in Training and Test data
    5530. +
    5531. Exercises
    5532. +
    5533. Exercise 1: Setting up various Python environments
    5534. +
    5535. Exercise 2: making your own data and exploring scikit-learn
    5536. +
    5537. Exercise 3: Split data in test and training data
    5538. @@ -351,24 +335,253 @@ MathJax.Hub.Config({

       

       

       

      -

      More Basic Matrix Features

      +

      Meet the Pandas

      -
      -
      - -

      For an \( N\times N \) matrix \( \mathbf{A} \) the following properties are all equivalent

      +

      +
      +

      +
      +

      -
        -
      • If the inverse of \( \mathbf{A} \) exists, \( \mathbf{A} \) is nonsingular.
      • -
      • The equation \( \mathbf{Ax}=0 \) implies \( \mathbf{x}=0 \).
      • -
      • The rows of \( \mathbf{A} \) form a basis of \( R^N \).
      • -
      • The columns of \( \mathbf{A} \) form a basis of \( R^N \).
      • -
      • \( \mathbf{A} \) is a product of elementary matrices.
      • -
      • \( 0 \) is not eigenvalue of \( \mathbf{A} \).
      • -
      +

      Another useful Python package is +pandas, which is an open source library +providing high-performance, easy-to-use data structures and data +analysis tools for Python. pandas stands for panel data, a term borrowed from econometrics and is an efficient library for data analysis with an emphasis on tabular data. +pandas has two major classes, the DataFrame class with two-dimensional data objects and tabular data organized in columns and the class Series with a focus on one-dimensional data objects. Both classes allow you to index data easily as we will see in the examples below. +pandas allows you also to perform mathematical operations on the data, spanning from simple reshapings of vectors and matrices to statistical operations. +

      + +

      The following simple example shows how we can, in an easy way make tables of our data. Here we define a data set which includes names, place of birth and date of birth, and displays the data in an easy to read way. We will see repeated use of pandas, in particular in connection with classification of data.

      + + + +
      +
      +
      +
      +
      +
      import pandas as pd
      +from IPython.display import display
      +data = {'First Name': ["Frodo", "Bilbo", "Aragorn II", "Samwise"],
      +        'Last Name': ["Baggins", "Baggins","Elessar","Gamgee"],
      +        'Place of birth': ["Shire", "Shire", "Eriador", "Shire"],
      +        'Date of Birth T.A.': [2968, 2890, 2931, 2980]
      +        }
      +data_pandas = pd.DataFrame(data)
      +display(data_pandas)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +

      In the above we have imported pandas with the shorthand pd, the latter has become the standard way we import pandas. We make then a list of various variables +and reorganize the aboves lists into a DataFrame and then print out a neat table with specific column labels as Name, place of birth and date of birth. +Displaying these results, we see that the indices are given by the default numbers from zero to three. +pandas is extremely flexible and we can easily change the above indices by defining a new type of indexing as +

      + + +
      +
      +
      +
      +
      +
      data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam'])
      +display(data_pandas)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Thereafter we display the content of the row which begins with the index Aragorn

      + + +
      +
      +
      +
      +
      +
      display(data_pandas.loc['Aragorn'])
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      We can easily append data to this, for example

      + + +
      +
      +
      +
      +
      +
      new_hobbit = {'First Name': ["Peregrin"],
      +              'Last Name': ["Took"],
      +              'Place of birth': ["Shire"],
      +              'Date of Birth T.A.': [2990]
      +              }
      +data_pandas=data_pandas.append(pd.DataFrame(new_hobbit, index=['Pippin']))
      +display(data_pandas)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Here are other examples where we use the DataFrame functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix +of dimensionality \( 10\times 5 \) and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations. +

      + + +
      +
      +
      +
      +
      +
      import numpy as np
      +import pandas as pd
      +from IPython.display import display
      +np.random.seed(100)
      +# setting up a 10 x 5 matrix
      +rows = 10
      +cols = 5
      +a = np.random.randn(rows,cols)
      +df = pd.DataFrame(a)
      +display(df)
      +print(df.mean())
      +print(df.std())
      +display(df**2)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Thereafter we can select specific columns only and plot final results

      + + +
      +
      +
      +
      +
      +
      df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']
      +df.index = np.arange(10)
      +
      +display(df)
      +print(df['Second'].mean() )
      +print(df.info())
      +print(df.describe())
      +
      +from pylab import plt, mpl
      +plt.style.use('seaborn')
      +mpl.rcParams['font.family'] = 'serif'
      +
      +df.cumsum().plot(lw=2.0, figsize=(10,6))
      +plt.show()
      +
      +
      +df.plot.bar(figsize=(10,6), rot=15)
      +plt.show()
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      We can produce a \( 4\times 4 \) matrix

      + + +
      +
      +
      +
      +
      +
      b = np.arange(16).reshape((4,4))
      +print(b)
      +df1 = pd.DataFrame(b)
      +print(df1)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      and many other operations.

      + +

      The Series class is another important class included in +pandas. You can view it as a specialization of DataFrame but where +we have just a single column of data. It shares many of the same features as DataFrame. As with DataFrame, +most operations are vectorized, achieving thereby a high performance when dealing with computations of arrays, in particular labeled arrays. +As we will see below it leads also to a very concice code close to the mathematical operations we may be interested in. +For multidimensional arrays, we recommend strongly xarray. xarray has much of the same flexibility as pandas, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both pandas and xarray. +

      @@ -395,7 +608,7 @@ MathJax.Hub.Config({

    5539. 41
    5540. 42
    5541. ...
    5542. -
    5543. 66
    5544. +
    5545. 62
    5546. »
    5547. diff --git a/doc/pub/week34/html/._week34-bs033.html b/doc/pub/week34/html/._week34-bs033.html index 0914f3c4f..b9d74b0c5 100644 --- a/doc/pub/week34/html/._week34-bs033.html +++ b/doc/pub/week34/html/._week34-bs033.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    5548. Installing R, C++, cython or Julia
    5549. Installing R, C++, cython, Numba etc
    5550. Numpy examples and Important Matrix and vector handling packages
    5551. -
    5552. Basic Matrix Features
    5553. -
    5554.    Some famous Matrices
    5555. -
    5556.    More Basic Matrix Features
    5557. -
    5558. Numpy and arrays
    5559. -
    5560. Matrices in Python
    5561. -
    5562. Meet the Pandas
    5563. -
    5564. Friday August 27
    5565. -
    5566.    Simple linear regression model using scikit-learn
    5567. -
    5568.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5569. -
    5570.    Organizing our data
    5571. -
    5572.    Seeing the wood for the trees
    5573. -
    5574.    And what about using neural networks?
    5575. -
    5576. A first summary
    5577. -
    5578. Why Linear Regression (aka Ordinary Least Squares and family)
    5579. -
    5580. Regression analysis, overarching aims
    5581. -
    5582. Regression analysis, overarching aims II
    5583. -
    5584. Examples
    5585. -
    5586. General linear models
    5587. -
    5588. Rewriting the fitting procedure as a linear algebra problem
    5589. -
    5590. Rewriting the fitting procedure as a linear algebra problem, more details
    5591. -
    5592. Generalizing the fitting procedure as a linear algebra problem
    5593. -
    5594. Generalizing the fitting procedure as a linear algebra problem
    5595. -
    5596. Optimizing our parameters
    5597. -
    5598. Our model for the nuclear binding energies
    5599. -
    5600. Optimizing our parameters, more details
    5601. -
    5602. Interpretations and optimizing our parameters
    5603. -
    5604. Interpretations and optimizing our parameters
    5605. -
    5606. Some useful matrix and vector expressions
    5607. -
    5608. Interpretations and optimizing our parameters
    5609. -
    5610. Own code for Ordinary Least Squares
    5611. -
    5612. Adding error analysis and training set up
    5613. -
    5614. The \( \chi^2 \) function
    5615. -
    5616. The \( \chi^2 \) function
    5617. -
    5618. The \( \chi^2 \) function
    5619. -
    5620. The \( \chi^2 \) function
    5621. -
    5622. The \( \chi^2 \) function
    5623. -
    5624. The \( \chi^2 \) function
    5625. -
    5626. Fitting an Equation of State for Dense Nuclear Matter
    5627. -
    5628. The code
    5629. -
    5630. Splitting our Data in Training and Test data
    5631. -
    5632. Exercises
    5633. -
    5634. Exercise 1: Setting up various Python environments
    5635. -
    5636. Exercise 2: making your own data and exploring scikit-learn
    5637. -
    5638. Exercise 3: Normalizing our data
    5639. +
    5640. Numpy and arrays
    5641. +
    5642. Matrices in Python
    5643. +
    5644. Meet the Pandas
    5645. +
    5646.    Simple linear regression model using scikit-learn
    5647. +
    5648.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5649. +
    5650.    Organizing our data
    5651. +
    5652.    And what about using neural networks?
    5653. +
    5654. A first summary
    5655. +
    5656. Why Linear Regression (aka Ordinary Least Squares and family)
    5657. +
    5658. Regression analysis, overarching aims
    5659. +
    5660. Regression analysis, overarching aims II
    5661. +
    5662. Examples
    5663. +
    5664. General linear models
    5665. +
    5666. Rewriting the fitting procedure as a linear algebra problem
    5667. +
    5668. Rewriting the fitting procedure as a linear algebra problem, more details
    5669. +
    5670. Generalizing the fitting procedure as a linear algebra problem
    5671. +
    5672. Generalizing the fitting procedure as a linear algebra problem
    5673. +
    5674. Optimizing our parameters
    5675. +
    5676. Our model for the nuclear binding energies
    5677. +
    5678. Optimizing our parameters, more details
    5679. +
    5680. Interpretations and optimizing our parameters
    5681. +
    5682. Interpretations and optimizing our parameters
    5683. +
    5684. Some useful matrix and vector expressions
    5685. +
    5686. Interpretations and optimizing our parameters
    5687. +
    5688. Own code for Ordinary Least Squares
    5689. +
    5690. Adding error analysis and training set up
    5691. +
    5692. The \( \chi^2 \) function
    5693. +
    5694. The \( \chi^2 \) function
    5695. +
    5696. The \( \chi^2 \) function
    5697. +
    5698. The \( \chi^2 \) function
    5699. +
    5700. The \( \chi^2 \) function
    5701. +
    5702. The \( \chi^2 \) function
    5703. +
    5704. Fitting an Equation of State for Dense Nuclear Matter
    5705. +
    5706. The code
    5707. +
    5708. Splitting our Data in Training and Test data
    5709. +
    5710. Exercises
    5711. +
    5712. Exercise 1: Setting up various Python environments
    5713. +
    5714. Exercise 2: making your own data and exploring scikit-learn
    5715. +
    5716. Exercise 3: Split data in test and training data
    5717. @@ -351,9 +335,168 @@ MathJax.Hub.Config({

       

       

       

      -

      Numpy and arrays

      -

      Numpy provides an easy way to handle arrays in Python. The standard way to import this library is as

      +

      Simple linear regression model using scikit-learn

      +

      We start with perhaps our simplest possible example, using Scikit-Learn to perform linear regression analysis on a data set produced by us.

      + +

      What follows is a simple Python code where we have defined a function +\( y \) in terms of the variable \( x \). Both are defined as vectors with \( 100 \) entries. +The numbers in the vector \( \boldsymbol{x} \) are given +by random numbers generated with a uniform distribution with entries +\( x_i \in [0,1] \) (more about probability distribution functions +later). These values are then used to define a function \( y(x) \) +(tabulated again as a vector) with a linear dependence on \( x \) plus a +random noise added via the normal distribution. +

      + +

      The Numpy functions are imported used the import numpy as np +statement and the random number generator for the uniform distribution +is called using the function np.random.rand(), where we specificy +that we want \( 100 \) random variables. Using Numpy we define +automatically an array with the specified number of elements, \( 100 \) in +our case. With the Numpy function randn() we can compute random +numbers with the normal distribution (mean value \( \mu \) equal to zero and +variance \( \sigma^2 \) set to one) and produce the values of \( y \) assuming a linear +dependence as function of \( x \) +

      + +$$ +y = 2x+N(0,1), +$$ + +

      where \( N(0,1) \) represents random numbers generated by the normal +distribution. From Scikit-Learn we import then the +LinearRegression functionality and make a prediction \( \tilde{y} = +\alpha + \beta x \) using the function fit(x,y). We call the set of +data \( (\boldsymbol{x},\boldsymbol{y}) \) for our training data. The Python package +scikit-learn has also a functionality which extracts the above +fitting parameters \( \alpha \) and \( \beta \) (see below). Later we will +distinguish between training data and test data. +

      + +

      For plotting we use the Python package +matplotlib which produces publication +quality figures. Feel free to explore the extensive +gallery of examples. In +this example we plot our original values of \( x \) and \( y \) as well as the +prediction ypredict (\( \tilde{y} \)), which attempts at fitting our +data with a straight line. +

      + +

      The Python code follows here.

      + + +
      +
      +
      +
      +
      +
      # Importing various packages
      +import numpy as np
      +import matplotlib.pyplot as plt
      +from sklearn.linear_model import LinearRegression
      +
      +x = np.random.rand(100,1)
      +y = 2*x+np.random.randn(100,1)
      +linreg = LinearRegression()
      +linreg.fit(x,y)
      +xnew = np.array([[0],[1]])
      +ypredict = linreg.predict(xnew)
      +
      +plt.plot(xnew, ypredict, "r-")
      +plt.plot(x, y ,'ro')
      +plt.axis([0,1.0,0, 5.0])
      +plt.xlabel(r'$x$')
      +plt.ylabel(r'$y$')
      +plt.title(r'Simple Linear Regression')
      +plt.show()
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      This example serves several aims. It allows us to demonstrate several +aspects of data analysis and later machine learning algorithms. The +immediate visualization shows that our linear fit is not +impressive. It goes through the data points, but there are many +outliers which are not reproduced by our linear regression. We could +now play around with this small program and change for example the +factor in front of \( x \) and the normal distribution. Try to change the +function \( y \) to +

      + +$$ +y = 10x+0.01 \times N(0,1), +$$ + +

      where \( x \) is defined as before. Does the fit look better? Indeed, by +reducing the role of the noise given by the normal distribution we see immediately that +our linear prediction seemingly reproduces better the training +set. However, this testing 'by the eye' is obviouly not satisfactory in the +long run. Here we have only defined the training data and our model, and +have not discussed a more rigorous approach to the cost function. +

      + +

      We need more rigorous criteria in defining whether we have succeeded or +not in modeling our training data. You will be surprised to see that +many scientists seldomly venture beyond this 'by the eye' approach. A +standard approach for the cost function is the so-called \( \chi^2 \) +function (a variant of the mean-squared error (MSE)) +

      + +$$ \chi^2 = \frac{1}{n} +\sum_{i=0}^{n-1}\frac{(y_i-\tilde{y}_i)^2}{\sigma_i^2}, +$$ + +

      where \( \sigma_i^2 \) is the variance (to be defined later) of the entry +\( y_i \). We may not know the explicit value of \( \sigma_i^2 \), it serves +however the aim of scaling the equations and make the cost function +dimensionless. +

      + +

      Minimizing the cost function is a central aspect of +our discussions to come. Finding its minima as function of the model +parameters (\( \alpha \) and \( \beta \) in our case) will be a recurring +theme in these series of lectures. Essentially all machine learning +algorithms we will discuss center around the minimization of the +chosen cost function. This depends in turn on our specific +model for describing the data, a typical situation in supervised +learning. Automatizing the search for the minima of the cost function is a +central ingredient in all algorithms. Typical methods which are +employed are various variants of gradient methods. These will be +discussed in more detail later. Again, you'll be surprised to hear that +many practitioners minimize the above function ''by the eye', popularly dubbed as +'chi by the eye'. That is, change a parameter and see (visually and numerically) that +the \( \chi^2 \) function becomes smaller. +

      + +

      There are many ways to define the cost function. A simpler approach is to look at the relative difference between the training data and the predicted data, that is we define +the relative error (why would we prefer the MSE instead of the relative error?) as +

      + +$$ +\epsilon_{\mathrm{relative}}= \frac{\vert \boldsymbol{y} -\boldsymbol{\tilde{y}}\vert}{\vert \boldsymbol{y}\vert}. +$$ + +

      The squared cost function results in an arithmetic mean-unbiased +estimator, and the absolute-value cost function results in a +median-unbiased estimator (in the one-dimensional case, and a +geometric median-unbiased estimator for the multi-dimensional +case). The squared cost function has the disadvantage that it has the tendency +to be dominated by outliers. +

      + +

      We can modify easily the above Python code and plot the relative error instead

      @@ -362,6 +505,21 @@ MathJax.Hub.Config({
      import numpy as np
      +import matplotlib.pyplot as plt
      +from sklearn.linear_model import LinearRegression
      +
      +x = np.random.rand(100,1)
      +y = 5*x+0.01*np.random.randn(100,1)
      +linreg = LinearRegression()
      +linreg.fit(x,y)
      +ypredict = linreg.predict(x)
      +
      +plt.plot(x, np.abs(ypredict-y)/abs(y), "ro")
      +plt.axis([0,1.0,0.0, 0.5])
      +plt.xlabel(r'$x$')
      +plt.ylabel(r'$\epsilon_{\mathrm{relative}}$')
      +plt.title(r'Relative error')
      +plt.show()
       
      @@ -377,34 +535,20 @@ MathJax.Hub.Config({
      -

      Here follows a simple example where we set up an array of ten elements, all determined by random numbers drawn according to the normal distribution,

      +

      Depending on the parameter in front of the normal distribution, we may +have a small or larger relative error. Try to play around with +different training data sets and study (graphically) the value of the +relative error. +

      - -
      -
      -
      -
      -
      -
      n = 10
      -x = np.random.normal(size=n)
      -print(x)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      +

      As mentioned above, Scikit-Learn has an impressive functionality. +We can for example extract the values of \( \alpha \) and \( \beta \) and +their error estimates, or the variance and standard deviation and many +other properties from the statistical data analysis. +

      -

      We defined a vector \( x \) with \( n=10 \) elements with its values given by the Normal distribution \( N(0,1) \). -Another alternative is to declare a vector as follows +

      Here we show an +example of the functionality of Scikit-Learn.

      @@ -413,9 +557,33 @@ Another alternative is to declare a vector as follows
      -
      import numpy as np
      -x = np.array([1, 2, 3])
      -print(x)
      +  
      import numpy as np 
      +import matplotlib.pyplot as plt 
      +from sklearn.linear_model import LinearRegression 
      +from sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error, mean_absolute_error
      +
      +x = np.random.rand(100,1)
      +y = 2.0+ 5*x+0.5*np.random.randn(100,1)
      +linreg = LinearRegression()
      +linreg.fit(x,y)
      +ypredict = linreg.predict(x)
      +print('The intercept alpha: \n', linreg.intercept_)
      +print('Coefficient beta : \n', linreg.coef_)
      +# The mean squared error                               
      +print("Mean squared error: %.2f" % mean_squared_error(y, ypredict))
      +# Explained variance score: 1 is perfect prediction                                 
      +print('Variance score: %.2f' % r2_score(y, ypredict))
      +# Mean squared log error                                                        
      +print('Mean squared log error: %.2f' % mean_squared_log_error(y, ypredict) )
      +# Mean absolute error                                                           
      +print('Mean absolute error: %.2f' % mean_absolute_error(y, ypredict))
      +plt.plot(x, ypredict, "r-")
      +plt.plot(x, y ,'ro')
      +plt.axis([0.0,1.0,1.5, 7.0])
      +plt.xlabel(r'$x$')
      +plt.ylabel(r'$y$')
      +plt.title(r'Linear Regression fit ')
      +plt.show()
       
      @@ -431,9 +599,159 @@ x = np.a
      -

      Here we have defined a vector with three elements, with \( x_0=1 \), \( x_1=2 \) and \( x_2=3 \). Note that both Python and C++ -start numbering array elements from \( 0 \) and on. This means that a vector with \( n \) elements has a sequence of entities \( x_0, x_1, x_2, \dots, x_{n-1} \). We could also let (recommended) Numpy to compute the logarithms of a specific array as +

      The function coef gives us the parameter \( \beta \) of our fit while intercept yields +\( \alpha \). Depending on the constant in front of the normal distribution, we get values near or far from \( \alpha =2 \) and \( \beta =5 \). Try to play around with different parameters in front of the normal distribution. The function meansquarederror gives us the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error or loss defined as

      +$$ MSE(\boldsymbol{y},\boldsymbol{\tilde{y}}) = \frac{1}{n} +\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, +$$ + +

      The smaller the value, the better the fit. Ideally we would like to +have an MSE equal zero. The attentive reader has probably recognized +this function as being similar to the \( \chi^2 \) function defined above. +

      + +

      The r2score function computes \( R^2 \), the coefficient of +determination. It provides a measure of how well future samples are +likely to be predicted by the model. Best possible score is 1.0 and it +can be negative (because the model can be arbitrarily worse). A +constant model that always predicts the expected value of \( \boldsymbol{y} \), +disregarding the input features, would get a \( R^2 \) score of \( 0.0 \). +

      + +

      If \( \tilde{\boldsymbol{y}}_i \) is the predicted value of the \( i-th \) sample and \( y_i \) is the corresponding true value, then the score \( R^2 \) is defined as

      +$$ +R^2(\boldsymbol{y}, \tilde{\boldsymbol{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, +$$ + +

      where we have defined the mean value of \( \boldsymbol{y} \) as

      +$$ +\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. +$$ + +

      Another quantity taht we will meet again in our discussions of regression analysis is + the mean absolute error (MAE), a risk metric corresponding to the expected value of the absolute error loss or what we call the \( l1 \)-norm loss. In our discussion above we presented the relative error. +The MAE is defined as follows +

      +$$ +\text{MAE}(\boldsymbol{y}, \boldsymbol{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n-1} \left| y_i - \tilde{y}_i \right|. +$$ + +

      We present the +squared logarithmic (quadratic) error +

      +$$ +\text{MSLE}(\boldsymbol{y}, \boldsymbol{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n - 1} (\log_e (1 + y_i) - \log_e (1 + \tilde{y}_i) )^2, +$$ + +

      where \( \log_e (x) \) stands for the natural logarithm of \( x \). This error +estimate is best to use when targets having exponential growth, such +as population counts, average sales of a commodity over a span of +years etc. +

      + +

      Finally, another cost function is the Huber cost function used in robust regression.

      + +

      The rationale behind this possible cost function is its reduced +sensitivity to outliers in the data set. In our discussions on +dimensionality reduction and normalization of data we will meet other +ways of dealing with outliers. +

      + +

      The Huber cost function is defined as

      +$$ +H_{\delta}(\boldsymbol{a})=\left\{\begin{array}{cc}\frac{1}{2} \boldsymbol{a}^{2}& \text{for }|\boldsymbol{a}|\leq \delta\\ \delta (|\boldsymbol{a}|-\frac{1}{2}\delta ),&\text{otherwise}.\end{array}\right. +$$ + +

      Here \( \boldsymbol{a}=\boldsymbol{y} - \boldsymbol{\tilde{y}} \).

      + +

      We will discuss in more detail these and other functions in the +various lectures and lab sessions. +

      +

      To our real data: nuclear binding energies. Brief reminder on masses and binding energies

      + +

      Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding +energies. A basic quantity which can be measured for the ground +states of nuclei is the atomic mass \( M(N, Z) \) of the neutral atom with +atomic mass number \( A \) and charge \( Z \). The number of neutrons is \( N \). There are indeed several sophisticated experiments worldwide which allow us to measure this quantity to high precision (parts per million even). +

      + +

      Atomic masses are usually tabulated in terms of the mass excess defined by

      +$$ +\Delta M(N, Z) = M(N, Z) - uA, +$$ + +

      where \( u \) is the Atomic Mass Unit

      +$$ +u = M(^{12}\mathrm{C})/12 = 931.4940954(57) \hspace{0.1cm} \mathrm{MeV}/c^2. +$$ + +

      The nucleon masses are

      +$$ +m_p = 1.00727646693(9)u, +$$ + +

      and

      +$$ +m_n = 939.56536(8)\hspace{0.1cm} \mathrm{MeV}/c^2 = 1.0086649156(6)u. +$$ + +

      In the 2016 mass evaluation of by W.J.Huang, G.Audi, M.Wang, F.G.Kondev, S.Naimi and X.Xu +there are data on masses and decays of 3437 nuclei. +

      + +

      The nuclear binding energy is defined as the energy required to break +up a given nucleus into its constituent parts of \( N \) neutrons and \( Z \) +protons. In terms of the atomic masses \( M(N, Z) \) the binding energy is +defined by +

      + +$$ +BE(N, Z) = ZM_H c^2 + Nm_n c^2 - M(N, Z)c^2 , +$$ + +

      where \( M_H \) is the mass of the hydrogen atom and \( m_n \) is the mass of the neutron. +In terms of the mass excess the binding energy is given by +

      +$$ +BE(N, Z) = Z\Delta_H c^2 + N\Delta_n c^2 -\Delta(N, Z)c^2 , +$$ + +

      where \( \Delta_H c^2 = 7.2890 \) MeV and \( \Delta_n c^2 = 8.0713 \) MeV.

      + +

      A popular and physically intuitive model which can be used to parametrize +the experimental binding energies as function of \( A \), is the so-called +liquid drop model. The ansatz is based on the following expression +

      + +$$ +BE(N,Z) = a_1A-a_2A^{2/3}-a_3\frac{Z^2}{A^{1/3}}-a_4\frac{(N-Z)^2}{A}, +$$ + +

      where \( A \) stands for the number of nucleons and the $a_i$s are parameters which are determined by a fit +to the experimental data. +

      + +

      To arrive at the above expression we have assumed that we can make the following assumptions:

      + +
        +
      • There is a volume term \( a_1A \) proportional with the number of nucleons (the energy is also an extensive quantity). When an assembly of nucleons of the same size is packed together into the smallest volume, each interior nucleon has a certain number of other nucleons in contact with it. This contribution is proportional to the volume.
      • +
      • There is a surface energy term \( a_2A^{2/3} \). The assumption here is that a nucleon at the surface of a nucleus interacts with fewer other nucleons than one in the interior of the nucleus and hence its binding energy is less. This surface energy term takes that into account and is therefore negative and is proportional to the surface area.
      • +
      • There is a Coulomb energy term \( a_3\frac{Z^2}{A^{1/3}} \). The electric repulsion between each pair of protons in a nucleus yields less binding.
      • +
      • There is an asymmetry term \( a_4\frac{(N-Z)^2}{A} \). This term is associated with the Pauli exclusion principle and reflects the fact that the proton-neutron interaction is more attractive on the average than the neutron-neutron and proton-proton interactions.
      • +
      +

      We could also add a so-called pairing term, which is a correction term that +arises from the tendency of proton pairs and neutron pairs to +occur. An even number of particles is more stable than an odd number. +

      +

      Organizing our data

      + +

      Let us start with reading and organizing our data. +We start with the compilation of masses and binding energies from 2016. +After having downloaded this file to our own computer, we are now ready to read the file and start structuring our data. +

      + +

      We start with preparing folders for storing our calculations and the data file over masses and binding energies. We import also various modules that we will find useful in order to present various Machine Learning methods. Here we focus mainly on the functionality of scikit-learn.

      @@ -441,9 +759,39 @@ start numbering array elements from \( 0 \) and on. This means that a vector wit
      -
      import numpy as np
      -x = np.log(np.array([4, 7, 8]))
      -print(x)
      +  
      # Common imports
      +import numpy as np
      +import pandas as pd
      +import matplotlib.pyplot as plt
      +import sklearn.linear_model as skl
      +from sklearn.model_selection import train_test_split
      +from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
      +import os
      +
      +# Where to save the figures and data files
      +PROJECT_ROOT_DIR = "Results"
      +FIGURE_ID = "Results/FigureFiles"
      +DATA_ID = "DataFiles/"
      +
      +if not os.path.exists(PROJECT_ROOT_DIR):
      +    os.mkdir(PROJECT_ROOT_DIR)
      +
      +if not os.path.exists(FIGURE_ID):
      +    os.makedirs(FIGURE_ID)
      +
      +if not os.path.exists(DATA_ID):
      +    os.makedirs(DATA_ID)
      +
      +def image_path(fig_id):
      +    return os.path.join(FIGURE_ID, fig_id)
      +
      +def data_path(dat_id):
      +    return os.path.join(DATA_ID, dat_id)
      +
      +def save_fig(fig_id):
      +    plt.savefig(image_path(fig_id) + ".png", format='png')
      +
      +infile = open(data_path("MassEval2016.dat"),'r')
       
      @@ -459,13 +807,84 @@ x = np.l
      -

      In the last example we used Numpy's unary function \( np.log \). This function is -highly tuned to compute array elements since the code is vectorized -and does not require looping. We normaly recommend that you use the -Numpy intrinsic functions instead of the corresponding log function -from Python's math module. The looping is done explicitely by the -np.log function. The alternative, and slower way to compute the -logarithms of a vector would be to write +

      Before we proceed, we define also a function for making our plots. You can obviously avoid this and simply set up various matplotlib commands every time you need them. You may however find it convenient to collect all such commands in one function and simply call this function.

      + + +
      +
      +
      +
      +
      +
      from pylab import plt, mpl
      +plt.style.use('seaborn')
      +mpl.rcParams['font.family'] = 'serif'
      +
      +def MakePlot(x,y, styles, labels, axlabels):
      +    plt.figure(figsize=(10,6))
      +    for i in range(len(x)):
      +        plt.plot(x[i], y[i], styles[i], label = labels[i])
      +        plt.xlabel(axlabels[0])
      +        plt.ylabel(axlabels[1])
      +    plt.legend(loc=0)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Our next step is to read the data on experimental binding energies and +reorganize them as functions of the mass number \( A \), the number of +protons \( Z \) and neutrons \( N \) using pandas. Before we do this it is +always useful (unless you have a binary file or other types of compressed +data) to actually open the file and simply take a look at it! +

      + +

      In particular, the program that outputs the final nuclear masses is written in Fortran with a specific format. It means that we need to figure out the format and which columns contain the data we are interested in. Pandas comes with a function that reads formatted output. After having admired the file, we are now ready to start massaging it with pandas. The file begins with some basic format information.

      + + +
      +
      +
      +
      +
      +
      """                                                                                                                         
      +This is taken from the data file of the mass 2016 evaluation.                                                               
      +All files are 3436 lines long with 124 character per line.                                                                  
      +       Headers are 39 lines long.                                                                                           
      +   col 1     :  Fortran character control: 1 = page feed  0 = line feed                                                     
      +   format    :  a1,i3,i5,i5,i5,1x,a3,a4,1x,f13.5,f11.5,f11.3,f9.3,1x,a2,f11.3,f9.3,1x,i3,1x,f12.5,f11.5                     
      +   These formats are reflected in the pandas widths variable below, see the statement                                       
      +   widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),                                                            
      +   Pandas has also a variable header, with length 39 in this case.                                                          
      +"""
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      The data we are interested in are in columns 2, 3, 4 and 11, giving us +the number of neutrons, protons, mass numbers and binding energies, +respectively. We add also for the sake of completeness the element name. The data are in fixed-width formatted lines and we will +covert them into the pandas DataFrame structure.

      @@ -475,12 +894,24 @@ logarithms of a vector would be to write
      -
      import numpy as np
      -from math import log
      -x = np.array([4, 7, 8])
      -for i in range(0, len(x)):
      -    x[i] = log(x[i])
      -print(x)
      +  
      # Read the experimental data with Pandas
      +Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),
      +              names=('N', 'Z', 'A', 'Element', 'Ebinding'),
      +              widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),
      +              header=39,
      +              index_col=False)
      +
      +# Extrapolated values are indicated by '#' in place of the decimal place, so
      +# the Ebinding column won't be numeric. Coerce to float and drop these entries.
      +Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')
      +Masses = Masses.dropna()
      +# Convert from keV to MeV.
      +Masses['Ebinding'] /= 1000
      +
      +# Group the DataFrame by nucleon number, A.
      +Masses = Masses.groupby('A')
      +# Find the rows of the grouped DataFrame with the maximum binding energy.
      +Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])
       
      @@ -496,8 +927,17 @@ x = np.a
      -

      We note that our code is much longer already and we need to import the log function from the math module. -The attentive reader will also notice that the output is \( [1, 1, 2] \). Python interprets automagically our numbers as integers (like the automatic keyword in C++). To change this we could define our array elements to be double precision numbers as +

      We have now read in the data, grouped them according to the variables we are interested in. +We see how easy it is to reorganize the data using pandas. If we +were to do these operations in C/C++ or Fortran, we would have had to +write various functions/subroutines which perform the above +reorganizations for us. Having reorganized the data, we can now start +to make some simple fits using both the functionalities in numpy and +Scikit-Learn afterwards. +

      + +

      Now we define five variables which contain +the number of nucleons \( A \), the number of protons \( Z \) and the number of neutrons \( N \), the element name and finally the energies themselves.

      @@ -506,9 +946,12 @@ The attentive reader will also notice that the output is \( [1, 1, 2] \). Python
      -
      import numpy as np
      -x = np.log(np.array([4, 7, 8], dtype = np.float64))
      -print(x)
      +  
      A = Masses['A']
      +Z = Masses['Z']
      +N = Masses['N']
      +Element = Masses['Element']
      +Energies = Masses['Ebinding']
      +print(Masses)
       
      @@ -524,7 +967,9 @@ x = np.l
      -

      or simply write them as double precision numbers (Python uses 64 bits as default for floating point type variables), that is

      +

      The next step, and we will define this mathematically later, is to set up the so-called design matrix. We will throughout call this matrix \( \boldsymbol{X} \). +It has dimensionality \( p\times n \), where \( n \) is the number of data points and \( p \) are the so-called predictors. In our case here they are given by the number of polynomials in \( A \) we wish to include in the fit. +

      @@ -532,9 +977,13 @@ x = np.l
      -
      import numpy as np
      -x = np.log(np.array([4.0, 7.0, 8.0]))
      -print(x)
      +  
      # Now we set up the design matrix X
      +X = np.zeros((len(A),5))
      +X[:,0] = 1
      +X[:,1] = A
      +X[:,2] = A**(2.0/3.0)
      +X[:,3] = A**(-1.0/3.0)
      +X[:,4] = A**(-1.0)
       
      @@ -550,7 +999,7 @@ x = np.l
      -

      To check the number of bytes (remember that one byte contains eight bits for double precision variables), you can use simple use the itemsize functionality (the array \( x \) is actually an object which inherits the functionalities defined in Numpy) as

      +

      With scikitlearn we are now ready to use linear regression and fit our data.

      @@ -558,9 +1007,8 @@ x = np.l
      -
      import numpy as np
      -x = np.log(np.array([4.0, 7.0, 8.0]))
      -print(x.itemsize)
      +  
      clf = skl.LinearRegression().fit(X, Energies)
      +fity = clf.predict(X)
       
      @@ -576,6 +1024,117 @@ x = np.l
      +

      Pretty simple! +Now we can print measures of how our fit is doing, the coefficients from the fits and plot the final fit together with our data. +

      + + +
      +
      +
      +
      +
      +
      # The mean squared error                               
      +print("Mean squared error: %.2f" % mean_squared_error(Energies, fity))
      +# Explained variance score: 1 is perfect prediction                                 
      +print('Variance score: %.2f' % r2_score(Energies, fity))
      +# Mean absolute error                                                           
      +print('Mean absolute error: %.2f' % mean_absolute_error(Energies, fity))
      +print(clf.coef_, clf.intercept_)
      +
      +Masses['Eapprox']  = fity
      +# Generate a plot comparing the experimental with the fitted values values.
      +fig, ax = plt.subplots()
      +ax.set_xlabel(r'$A = N + Z$')
      +ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$')
      +ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,
      +            label='Ame2016')
      +ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',
      +            label='Fit')
      +ax.legend()
      +save_fig("Masses2016")
      +plt.show()
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +

      And what about using neural networks?

      +

      The seaborn package allows us to visualize data in an efficient way. Note that we use scikit-learn's multi-layer perceptron (or feed forward neural network) +functionality. +

      + + +
      +
      +
      +
      +
      +
      from sklearn.neural_network import MLPRegressor
      +from sklearn.metrics import accuracy_score
      +import seaborn as sns
      +
      +X_train = X
      +Y_train = Energies
      +n_hidden_neurons = 100
      +epochs = 100
      +# store models for later use
      +eta_vals = np.logspace(-5, 1, 7)
      +lmbd_vals = np.logspace(-5, 1, 7)
      +# store the models for later use
      +DNN_scikit = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
      +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
      +sns.set()
      +for i, eta in enumerate(eta_vals):
      +    for j, lmbd in enumerate(lmbd_vals):
      +        dnn = MLPRegressor(hidden_layer_sizes=(n_hidden_neurons), activation='logistic',
      +                            alpha=lmbd, learning_rate_init=eta, max_iter=epochs)
      +        dnn.fit(X_train, Y_train)
      +        DNN_scikit[i][j] = dnn
      +        train_accuracy[i][j] = dnn.score(X_train, Y_train)
      +
      +fig, ax = plt.subplots(figsize = (10, 10))
      +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
      +ax.set_title("Training Accuracy")
      +ax.set_ylabel("$\eta$")
      +ax.set_xlabel("$\lambda$")
      +plt.show()
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +

      A first summary

      + +

      The aim behind these introductory words was to present to you various +Python libraries and their functionalities, in particular libraries like +numpy, pandas, xarray and matplotlib and other that make our life much easier +in handling various data sets and visualizing data. +

      + +

      Furthermore, +Scikit-Learn allows us with few lines of code to implement popular +Machine Learning algorithms for supervised learning. Later we will meet Tensorflow, a powerful library for deep learning. +Now it is time to dive more into the details of various methods. We will start with linear regression and try to take a deeper look at what it entails. +

      @@ -602,7 +1161,7 @@ x = np.l

    5718. 42
    5719. 43
    5720. ...
    5721. -
    5722. 66
    5723. +
    5724. 62
    5725. »
    5726. diff --git a/doc/pub/week34/html/._week34-bs034.html b/doc/pub/week34/html/._week34-bs034.html index dce995f67..ce6e498af 100644 --- a/doc/pub/week34/html/._week34-bs034.html +++ b/doc/pub/week34/html/._week34-bs034.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    5727. Installing R, C++, cython or Julia
    5728. Installing R, C++, cython, Numba etc
    5729. Numpy examples and Important Matrix and vector handling packages
    5730. -
    5731. Basic Matrix Features
    5732. -
    5733.    Some famous Matrices
    5734. -
    5735.    More Basic Matrix Features
    5736. -
    5737. Numpy and arrays
    5738. -
    5739. Matrices in Python
    5740. -
    5741. Meet the Pandas
    5742. -
    5743. Friday August 27
    5744. -
    5745.    Simple linear regression model using scikit-learn
    5746. -
    5747.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5748. -
    5749.    Organizing our data
    5750. -
    5751.    Seeing the wood for the trees
    5752. -
    5753.    And what about using neural networks?
    5754. -
    5755. A first summary
    5756. -
    5757. Why Linear Regression (aka Ordinary Least Squares and family)
    5758. -
    5759. Regression analysis, overarching aims
    5760. -
    5761. Regression analysis, overarching aims II
    5762. -
    5763. Examples
    5764. -
    5765. General linear models
    5766. -
    5767. Rewriting the fitting procedure as a linear algebra problem
    5768. -
    5769. Rewriting the fitting procedure as a linear algebra problem, more details
    5770. -
    5771. Generalizing the fitting procedure as a linear algebra problem
    5772. -
    5773. Generalizing the fitting procedure as a linear algebra problem
    5774. -
    5775. Optimizing our parameters
    5776. -
    5777. Our model for the nuclear binding energies
    5778. -
    5779. Optimizing our parameters, more details
    5780. -
    5781. Interpretations and optimizing our parameters
    5782. -
    5783. Interpretations and optimizing our parameters
    5784. -
    5785. Some useful matrix and vector expressions
    5786. -
    5787. Interpretations and optimizing our parameters
    5788. -
    5789. Own code for Ordinary Least Squares
    5790. -
    5791. Adding error analysis and training set up
    5792. -
    5793. The \( \chi^2 \) function
    5794. -
    5795. The \( \chi^2 \) function
    5796. -
    5797. The \( \chi^2 \) function
    5798. -
    5799. The \( \chi^2 \) function
    5800. -
    5801. The \( \chi^2 \) function
    5802. -
    5803. The \( \chi^2 \) function
    5804. -
    5805. Fitting an Equation of State for Dense Nuclear Matter
    5806. -
    5807. The code
    5808. -
    5809. Splitting our Data in Training and Test data
    5810. -
    5811. Exercises
    5812. -
    5813. Exercise 1: Setting up various Python environments
    5814. -
    5815. Exercise 2: making your own data and exploring scikit-learn
    5816. -
    5817. Exercise 3: Normalizing our data
    5818. +
    5819. Numpy and arrays
    5820. +
    5821. Matrices in Python
    5822. +
    5823. Meet the Pandas
    5824. +
    5825.    Simple linear regression model using scikit-learn
    5826. +
    5827.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5828. +
    5829.    Organizing our data
    5830. +
    5831.    And what about using neural networks?
    5832. +
    5833. A first summary
    5834. +
    5835. Why Linear Regression (aka Ordinary Least Squares and family)
    5836. +
    5837. Regression analysis, overarching aims
    5838. +
    5839. Regression analysis, overarching aims II
    5840. +
    5841. Examples
    5842. +
    5843. General linear models
    5844. +
    5845. Rewriting the fitting procedure as a linear algebra problem
    5846. +
    5847. Rewriting the fitting procedure as a linear algebra problem, more details
    5848. +
    5849. Generalizing the fitting procedure as a linear algebra problem
    5850. +
    5851. Generalizing the fitting procedure as a linear algebra problem
    5852. +
    5853. Optimizing our parameters
    5854. +
    5855. Our model for the nuclear binding energies
    5856. +
    5857. Optimizing our parameters, more details
    5858. +
    5859. Interpretations and optimizing our parameters
    5860. +
    5861. Interpretations and optimizing our parameters
    5862. +
    5863. Some useful matrix and vector expressions
    5864. +
    5865. Interpretations and optimizing our parameters
    5866. +
    5867. Own code for Ordinary Least Squares
    5868. +
    5869. Adding error analysis and training set up
    5870. +
    5871. The \( \chi^2 \) function
    5872. +
    5873. The \( \chi^2 \) function
    5874. +
    5875. The \( \chi^2 \) function
    5876. +
    5877. The \( \chi^2 \) function
    5878. +
    5879. The \( \chi^2 \) function
    5880. +
    5881. The \( \chi^2 \) function
    5882. +
    5883. Fitting an Equation of State for Dense Nuclear Matter
    5884. +
    5885. The code
    5886. +
    5887. Splitting our Data in Training and Test data
    5888. +
    5889. Exercises
    5890. +
    5891. Exercise 1: Setting up various Python environments
    5892. +
    5893. Exercise 2: making your own data and exploring scikit-learn
    5894. +
    5895. Exercise 3: Split data in test and training data
    5896. @@ -351,277 +335,24 @@ MathJax.Hub.Config({

       

       

       

      -

      Matrices in Python

      +

      Why Linear Regression (aka Ordinary Least Squares and family)

      -

      Having defined vectors, we are now ready to try out matrices. We can -define a \( 3 \times 3 \) real matrix \( \boldsymbol{A} \) as (recall that we user -lowercase letters for vectors and uppercase letters for matrices) +

      Fitting a continuous function with linear parameterization in terms of the parameters \( \boldsymbol{\beta} \).

      +
        +
      • Method of choice for fitting a continuous function!
      • +
      • Gives an excellent introduction to central Machine Learning features with understandable pedagogical links to other methods like Neural Networks, Support Vector Machines etc
      • +
      • Analytical expression for the fitting parameters \( \boldsymbol{\beta} \)
      • +
      • Analytical expressions for statistical propertiers like mean values, variances, confidence intervals and more
      • +
      • Analytical relation with probabilistic interpretations
      • +
      • Easy to introduce basic concepts like bias-variance tradeoff, cross-validation, resampling and regularization techniques and many other ML topics
      • +
      • Easy to code! And links well with classification problems and logistic regression and neural networks
      • +
      • Allows for easy hands-on understanding of gradient descent methods
      • +
      • and many more features
      • +
      +

      For more discussions of Ridge and Lasso regression, Wessel van Wieringen's article is highly recommended. +Similarly, Mehta et al's article is also recommended.

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
      -print(A)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      If we use the shape function we would get \( (3, 3) \) as output, that is verifying that our matrix is a \( 3\times 3 \) matrix. We can slice the matrix and print for example the first column (Python organized matrix elements in a row-major order, see below) as

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
      -# print the first column, row-major order and elements start with 0
      -print(A[:,0]) 
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      We can continue this was by printing out other columns or rows. The example here prints out the second column

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
      -# print the first column, row-major order and elements start with 0
      -print(A[1,:]) 
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Numpy contains many other functionalities that allow us to slice, subdivide etc etc arrays. We strongly recommend that you look up the Numpy website for more details. Useful functions when defining a matrix are the np.zeros function which declares a matrix of a given dimension and sets all elements to zero

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -n = 10
      -# define a matrix of dimension 10 x 10 and set all elements to zero
      -A = np.zeros( (n, n) )
      -print(A) 
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      or initializing all elements to

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -n = 10
      -# define a matrix of dimension 10 x 10 and set all elements to one
      -A = np.ones( (n, n) )
      -print(A) 
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      or as unitarily distributed random numbers (see the material on random number generators in the statistics part)

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -n = 10
      -# define a matrix of dimension 10 x 10 and set all elements to random numbers with x \in [0, 1]
      -A = np.random.rand(n, n)
      -print(A) 
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      As we will see throughout these lectures, there are several extremely useful functionalities in Numpy. -As an example, consider the discussion of the covariance matrix. Suppose we have defined three vectors -\( \boldsymbol{x}, \boldsymbol{y}, \boldsymbol{z} \) with \( n \) elements each. The covariance matrix is defined as -

      -$$ -\boldsymbol{\Sigma} = \begin{bmatrix} \sigma_{xx} & \sigma_{xy} & \sigma_{xz} \\ - \sigma_{yx} & \sigma_{yy} & \sigma_{yz} \\ - \sigma_{zx} & \sigma_{zy} & \sigma_{zz} - \end{bmatrix}, -$$ - -

      where for example

      -$$ -\sigma_{xy} =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). -$$ - -

      The Numpy function np.cov calculates the covariance elements using the factor \( 1/(n-1) \) instead of \( 1/n \) since it assumes we do not have the exact mean values. -The following simple function uses the np.vstack function which takes each vector of dimension \( 1\times n \) and produces a \( 3\times n \) matrix \( \boldsymbol{W} \) -

      -$$ -\boldsymbol{W} = \begin{bmatrix} x_0 & x_1 & x_2 & \dots & x_{n-2} & x_{n-1} \\ - y_0 & y_1 & y_2 & \dots & y_{n-2} & y_{n-1} \\ - z_0 & z_1 & z_2 & \dots & z_{n-2} & z_{n-1} \\ - \end{bmatrix}, -$$ - -

      which in turn is converted into into the \( 3\times 3 \) covariance matrix -\( \boldsymbol{\Sigma} \) via the Numpy function np.cov(). We note that we can also calculate -the mean value of each set of samples \( \boldsymbol{x} \) etc using the Numpy -function np.mean(x). We can also extract the eigenvalues of the -covariance matrix through the np.linalg.eig() function. -

      - - - -
      -
      -
      -
      -
      -
      # Importing various packages
      -import numpy as np
      -
      -n = 100
      -x = np.random.normal(size=n)
      -print(np.mean(x))
      -y = 4+3*x+np.random.normal(size=n)
      -print(np.mean(y))
      -z = x**3+np.random.normal(size=n)
      -print(np.mean(z))
      -W = np.vstack((x, y, z))
      -Sigma = np.cov(W)
      -print(Sigma)
      -Eigvals, Eigvecs = np.linalg.eig(Sigma)
      -print(Eigvals)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -
      -
      -
      -
      -
      -
      import numpy as np
      -import matplotlib.pyplot as plt
      -from scipy import sparse
      -eye = np.eye(4)
      -print(eye)
      -sparse_mtx = sparse.csr_matrix(eye)
      -print(sparse_mtx)
      -x = np.linspace(-10,10,100)
      -y = np.sin(x)
      -plt.plot(x,y,marker='x')
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      diff --git a/doc/pub/week34/html/._week34-bs035.html b/doc/pub/week34/html/._week34-bs035.html index 8b950918b..31dbbae4d 100644 --- a/doc/pub/week34/html/._week34-bs035.html +++ b/doc/pub/week34/html/._week34-bs035.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    5897. Installing R, C++, cython or Julia
    5898. Installing R, C++, cython, Numba etc
    5899. Numpy examples and Important Matrix and vector handling packages
    5900. -
    5901. Basic Matrix Features
    5902. -
    5903.    Some famous Matrices
    5904. -
    5905.    More Basic Matrix Features
    5906. -
    5907. Numpy and arrays
    5908. -
    5909. Matrices in Python
    5910. -
    5911. Meet the Pandas
    5912. -
    5913. Friday August 27
    5914. -
    5915.    Simple linear regression model using scikit-learn
    5916. -
    5917.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5918. -
    5919.    Organizing our data
    5920. -
    5921.    Seeing the wood for the trees
    5922. -
    5923.    And what about using neural networks?
    5924. -
    5925. A first summary
    5926. -
    5927. Why Linear Regression (aka Ordinary Least Squares and family)
    5928. -
    5929. Regression analysis, overarching aims
    5930. -
    5931. Regression analysis, overarching aims II
    5932. -
    5933. Examples
    5934. -
    5935. General linear models
    5936. -
    5937. Rewriting the fitting procedure as a linear algebra problem
    5938. -
    5939. Rewriting the fitting procedure as a linear algebra problem, more details
    5940. -
    5941. Generalizing the fitting procedure as a linear algebra problem
    5942. -
    5943. Generalizing the fitting procedure as a linear algebra problem
    5944. -
    5945. Optimizing our parameters
    5946. -
    5947. Our model for the nuclear binding energies
    5948. -
    5949. Optimizing our parameters, more details
    5950. -
    5951. Interpretations and optimizing our parameters
    5952. -
    5953. Interpretations and optimizing our parameters
    5954. -
    5955. Some useful matrix and vector expressions
    5956. -
    5957. Interpretations and optimizing our parameters
    5958. -
    5959. Own code for Ordinary Least Squares
    5960. -
    5961. Adding error analysis and training set up
    5962. -
    5963. The \( \chi^2 \) function
    5964. -
    5965. The \( \chi^2 \) function
    5966. -
    5967. The \( \chi^2 \) function
    5968. -
    5969. The \( \chi^2 \) function
    5970. -
    5971. The \( \chi^2 \) function
    5972. -
    5973. The \( \chi^2 \) function
    5974. -
    5975. Fitting an Equation of State for Dense Nuclear Matter
    5976. -
    5977. The code
    5978. -
    5979. Splitting our Data in Training and Test data
    5980. -
    5981. Exercises
    5982. -
    5983. Exercise 1: Setting up various Python environments
    5984. -
    5985. Exercise 2: making your own data and exploring scikit-learn
    5986. -
    5987. Exercise 3: Normalizing our data
    5988. +
    5989. Numpy and arrays
    5990. +
    5991. Matrices in Python
    5992. +
    5993. Meet the Pandas
    5994. +
    5995.    Simple linear regression model using scikit-learn
    5996. +
    5997.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    5998. +
    5999.    Organizing our data
    6000. +
    6001.    And what about using neural networks?
    6002. +
    6003. A first summary
    6004. +
    6005. Why Linear Regression (aka Ordinary Least Squares and family)
    6006. +
    6007. Regression analysis, overarching aims
    6008. +
    6009. Regression analysis, overarching aims II
    6010. +
    6011. Examples
    6012. +
    6013. General linear models
    6014. +
    6015. Rewriting the fitting procedure as a linear algebra problem
    6016. +
    6017. Rewriting the fitting procedure as a linear algebra problem, more details
    6018. +
    6019. Generalizing the fitting procedure as a linear algebra problem
    6020. +
    6021. Generalizing the fitting procedure as a linear algebra problem
    6022. +
    6023. Optimizing our parameters
    6024. +
    6025. Our model for the nuclear binding energies
    6026. +
    6027. Optimizing our parameters, more details
    6028. +
    6029. Interpretations and optimizing our parameters
    6030. +
    6031. Interpretations and optimizing our parameters
    6032. +
    6033. Some useful matrix and vector expressions
    6034. +
    6035. Interpretations and optimizing our parameters
    6036. +
    6037. Own code for Ordinary Least Squares
    6038. +
    6039. Adding error analysis and training set up
    6040. +
    6041. The \( \chi^2 \) function
    6042. +
    6043. The \( \chi^2 \) function
    6044. +
    6045. The \( \chi^2 \) function
    6046. +
    6047. The \( \chi^2 \) function
    6048. +
    6049. The \( \chi^2 \) function
    6050. +
    6051. The \( \chi^2 \) function
    6052. +
    6053. Fitting an Equation of State for Dense Nuclear Matter
    6054. +
    6055. The code
    6056. +
    6057. Splitting our Data in Training and Test data
    6058. +
    6059. Exercises
    6060. +
    6061. Exercise 1: Setting up various Python environments
    6062. +
    6063. Exercise 2: making your own data and exploring scikit-learn
    6064. +
    6065. Exercise 3: Split data in test and training data
    6066. @@ -351,253 +335,25 @@ MathJax.Hub.Config({

       

       

       

      -

      Meet the Pandas

      +

      Regression analysis, overarching aims

      +
      +
      + -

      -
      -

      -
      -

      - -

      Another useful Python package is -pandas, which is an open source library -providing high-performance, easy-to-use data structures and data -analysis tools for Python. pandas stands for panel data, a term borrowed from econometrics and is an efficient library for data analysis with an emphasis on tabular data. -pandas has two major classes, the DataFrame class with two-dimensional data objects and tabular data organized in columns and the class Series with a focus on one-dimensional data objects. Both classes allow you to index data easily as we will see in the examples below. -pandas allows you also to perform mathematical operations on the data, spanning from simple reshapings of vectors and matrices to statistical operations. +

      Regression modeling deals with the description of the sampling distribution of a given random variable \( y \) and how it varies as function of another variable or a set of such variables \( \boldsymbol{x} =[x_0, x_1,\dots, x_{n-1}]^T \). +The first variable is called the dependent, the outcome or the response variable while the set of variables \( \boldsymbol{x} \) is called the independent variable, or the predictor variable or the explanatory variable.

      -

      The following simple example shows how we can, in an easy way make tables of our data. Here we define a data set which includes names, place of birth and date of birth, and displays the data in an easy to read way. We will see repeated use of pandas, in particular in connection with classification of data.

      - - - -
      -
      -
      -
      -
      -
      import pandas as pd
      -from IPython.display import display
      -data = {'First Name': ["Frodo", "Bilbo", "Aragorn II", "Samwise"],
      -        'Last Name': ["Baggins", "Baggins","Elessar","Gamgee"],
      -        'Place of birth': ["Shire", "Shire", "Eriador", "Shire"],
      -        'Date of Birth T.A.': [2968, 2890, 2931, 2980]
      -        }
      -data_pandas = pd.DataFrame(data)
      -display(data_pandas)
      -
      +

      A regression model aims at finding a likelihood function \( p(\boldsymbol{y}\vert \boldsymbol{x}) \), that is the conditional distribution for \( \boldsymbol{y} \) with a given \( \boldsymbol{x} \). The estimation of \( p(\boldsymbol{y}\vert \boldsymbol{x}) \) is made using a data set with

      +
        +
      • \( n \) cases \( i = 0, 1, 2, \dots, n-1 \)
      • +
      • Response (target, dependent or outcome) variable \( y_i \) with \( i = 0, 1, 2, \dots, n-1 \)
      • +
      • \( p \) so-called explanatory (independent or predictor) variables \( \boldsymbol{x}_i=[x_{i0}, x_{i1}, \dots, x_{ip-1}] \) with \( i = 0, 1, 2, \dots, n-1 \) and explanatory variables running from \( 0 \) to \( p-1 \). See below for more explicit examples.
      • +
      +

      The goal of the regression analysis is to extract/exploit relationship between \( \boldsymbol{y} \) and \( \boldsymbol{x} \) in or to infer causal dependencies, approximations to the likelihood functions, functional relationships and to make predictions, making fits and many other things.

      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      In the above we have imported pandas with the shorthand pd, the latter has become the standard way we import pandas. We make then a list of various variables -and reorganize the aboves lists into a DataFrame and then print out a neat table with specific column labels as Name, place of birth and date of birth. -Displaying these results, we see that the indices are given by the default numbers from zero to three. -pandas is extremely flexible and we can easily change the above indices by defining a new type of indexing as -

      - - -
      -
      -
      -
      -
      -
      data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam'])
      -display(data_pandas)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Thereafter we display the content of the row which begins with the index Aragorn

      - - -
      -
      -
      -
      -
      -
      display(data_pandas.loc['Aragorn'])
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      We can easily append data to this, for example

      - - -
      -
      -
      -
      -
      -
      new_hobbit = {'First Name': ["Peregrin"],
      -              'Last Name': ["Took"],
      -              'Place of birth': ["Shire"],
      -              'Date of Birth T.A.': [2990]
      -              }
      -data_pandas=data_pandas.append(pd.DataFrame(new_hobbit, index=['Pippin']))
      -display(data_pandas)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Here are other examples where we use the DataFrame functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix -of dimensionality \( 10\times 5 \) and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations. -

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -import pandas as pd
      -from IPython.display import display
      -np.random.seed(100)
      -# setting up a 10 x 5 matrix
      -rows = 10
      -cols = 5
      -a = np.random.randn(rows,cols)
      -df = pd.DataFrame(a)
      -display(df)
      -print(df.mean())
      -print(df.std())
      -display(df**2)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Thereafter we can select specific columns only and plot final results

      - - -
      -
      -
      -
      -
      -
      df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']
      -df.index = np.arange(10)
      -
      -display(df)
      -print(df['Second'].mean() )
      -print(df.info())
      -print(df.describe())
      -
      -from pylab import plt, mpl
      -plt.style.use('seaborn')
      -mpl.rcParams['font.family'] = 'serif'
      -
      -df.cumsum().plot(lw=2.0, figsize=(10,6))
      -plt.show()
      -
      -
      -df.plot.bar(figsize=(10,6), rot=15)
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      We can produce a \( 4\times 4 \) matrix

      - - -
      -
      -
      -
      -
      -
      b = np.arange(16).reshape((4,4))
      -print(b)
      -df1 = pd.DataFrame(b)
      -print(df1)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      and many other operations.

      - -

      The Series class is another important class included in -pandas. You can view it as a specialization of DataFrame but where -we have just a single column of data. It shares many of the same features as DataFrame. As with DataFrame, -most operations are vectorized, achieving thereby a high performance when dealing with computations of arrays, in particular labeled arrays. -As we will see below it leads also to a very concice code close to the mathematical operations we may be interested in. -For multidimensional arrays, we recommend strongly xarray. xarray has much of the same flexibility as pandas, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both pandas and xarray. -

      @@ -624,7 +380,7 @@ For multidimensional arrays, we recommend strongly 44

    6067. 45
    6068. ...
    6069. -
    6070. 66
    6071. +
    6072. 62
    6073. »
    6074. diff --git a/doc/pub/week34/html/._week34-bs036.html b/doc/pub/week34/html/._week34-bs036.html index e16415e53..cbf6191f6 100644 --- a/doc/pub/week34/html/._week34-bs036.html +++ b/doc/pub/week34/html/._week34-bs036.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    6075. Installing R, C++, cython or Julia
    6076. Installing R, C++, cython, Numba etc
    6077. Numpy examples and Important Matrix and vector handling packages
    6078. -
    6079. Basic Matrix Features
    6080. -
    6081.    Some famous Matrices
    6082. -
    6083.    More Basic Matrix Features
    6084. -
    6085. Numpy and arrays
    6086. -
    6087. Matrices in Python
    6088. -
    6089. Meet the Pandas
    6090. -
    6091. Friday August 27
    6092. -
    6093.    Simple linear regression model using scikit-learn
    6094. -
    6095.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6096. -
    6097.    Organizing our data
    6098. -
    6099.    Seeing the wood for the trees
    6100. -
    6101.    And what about using neural networks?
    6102. -
    6103. A first summary
    6104. -
    6105. Why Linear Regression (aka Ordinary Least Squares and family)
    6106. -
    6107. Regression analysis, overarching aims
    6108. -
    6109. Regression analysis, overarching aims II
    6110. -
    6111. Examples
    6112. -
    6113. General linear models
    6114. -
    6115. Rewriting the fitting procedure as a linear algebra problem
    6116. -
    6117. Rewriting the fitting procedure as a linear algebra problem, more details
    6118. -
    6119. Generalizing the fitting procedure as a linear algebra problem
    6120. -
    6121. Generalizing the fitting procedure as a linear algebra problem
    6122. -
    6123. Optimizing our parameters
    6124. -
    6125. Our model for the nuclear binding energies
    6126. -
    6127. Optimizing our parameters, more details
    6128. -
    6129. Interpretations and optimizing our parameters
    6130. -
    6131. Interpretations and optimizing our parameters
    6132. -
    6133. Some useful matrix and vector expressions
    6134. -
    6135. Interpretations and optimizing our parameters
    6136. -
    6137. Own code for Ordinary Least Squares
    6138. -
    6139. Adding error analysis and training set up
    6140. -
    6141. The \( \chi^2 \) function
    6142. -
    6143. The \( \chi^2 \) function
    6144. -
    6145. The \( \chi^2 \) function
    6146. -
    6147. The \( \chi^2 \) function
    6148. -
    6149. The \( \chi^2 \) function
    6150. -
    6151. The \( \chi^2 \) function
    6152. -
    6153. Fitting an Equation of State for Dense Nuclear Matter
    6154. -
    6155. The code
    6156. -
    6157. Splitting our Data in Training and Test data
    6158. -
    6159. Exercises
    6160. -
    6161. Exercise 1: Setting up various Python environments
    6162. -
    6163. Exercise 2: making your own data and exploring scikit-learn
    6164. -
    6165. Exercise 3: Normalizing our data
    6166. +
    6167. Numpy and arrays
    6168. +
    6169. Matrices in Python
    6170. +
    6171. Meet the Pandas
    6172. +
    6173.    Simple linear regression model using scikit-learn
    6174. +
    6175.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6176. +
    6177.    Organizing our data
    6178. +
    6179.    And what about using neural networks?
    6180. +
    6181. A first summary
    6182. +
    6183. Why Linear Regression (aka Ordinary Least Squares and family)
    6184. +
    6185. Regression analysis, overarching aims
    6186. +
    6187. Regression analysis, overarching aims II
    6188. +
    6189. Examples
    6190. +
    6191. General linear models
    6192. +
    6193. Rewriting the fitting procedure as a linear algebra problem
    6194. +
    6195. Rewriting the fitting procedure as a linear algebra problem, more details
    6196. +
    6197. Generalizing the fitting procedure as a linear algebra problem
    6198. +
    6199. Generalizing the fitting procedure as a linear algebra problem
    6200. +
    6201. Optimizing our parameters
    6202. +
    6203. Our model for the nuclear binding energies
    6204. +
    6205. Optimizing our parameters, more details
    6206. +
    6207. Interpretations and optimizing our parameters
    6208. +
    6209. Interpretations and optimizing our parameters
    6210. +
    6211. Some useful matrix and vector expressions
    6212. +
    6213. Interpretations and optimizing our parameters
    6214. +
    6215. Own code for Ordinary Least Squares
    6216. +
    6217. Adding error analysis and training set up
    6218. +
    6219. The \( \chi^2 \) function
    6220. +
    6221. The \( \chi^2 \) function
    6222. +
    6223. The \( \chi^2 \) function
    6224. +
    6225. The \( \chi^2 \) function
    6226. +
    6227. The \( \chi^2 \) function
    6228. +
    6229. The \( \chi^2 \) function
    6230. +
    6231. Fitting an Equation of State for Dense Nuclear Matter
    6232. +
    6233. The code
    6234. +
    6235. Splitting our Data in Training and Test data
    6236. +
    6237. Exercises
    6238. +
    6239. Exercise 1: Setting up various Python environments
    6240. +
    6241. Exercise 2: making your own data and exploring scikit-learn
    6242. +
    6243. Exercise 3: Split data in test and training data
    6244. @@ -351,11 +335,33 @@ MathJax.Hub.Config({

       

       

       

      -

      Friday August 27

      +

      Regression analysis, overarching aims II

      +
      +
      + -

      "Video of Lecture August 27, 2021":"https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h21/forelesningsvideoer/LectureThursdayAugust27.mp4?vrtx=view-as-webpage

      +

      Consider an experiment in which \( p \) characteristics of \( n \) samples are +measured. The data from this experiment, for various explanatory variables \( p \) are normally represented by a matrix +\( \mathbf{X} \). +

      + +

      The matrix \( \mathbf{X} \) is called the design +matrix. Additional information of the samples is available in the +form of \( \boldsymbol{y} \) (also as above). The variable \( \boldsymbol{y} \) is +generally referred to as the response variable. The aim of +regression analysis is to explain \( \boldsymbol{y} \) in terms of +\( \boldsymbol{X} \) through a functional relationship like \( y_i = +f(\mathbf{X}_{i,\ast}) \). When no prior knowledge on the form of +\( f(\cdot) \) is available, it is common to assume a linear relationship +between \( \boldsymbol{X} \) and \( \boldsymbol{y} \). This assumption gives rise to +the linear regression model where \( \boldsymbol{\beta} = [\beta_0, \ldots, +\beta_{p-1}]^{T} \) are the regression parameters. +

      + +

      Linear regression gives us a set of analytical equations for the parameters \( \beta_j \).

      +
      +
      -

      Video of Lecture from fall 2020 and Handwritten notes

      @@ -382,7 +388,7 @@ MathJax.Hub.Config({

    6245. 45
    6246. 46
    6247. ...
    6248. -
    6249. 66
    6250. +
    6251. 62
    6252. »
    6253. diff --git a/doc/pub/week34/html/._week34-bs037.html b/doc/pub/week34/html/._week34-bs037.html index 7e4fcd790..d4b7b24fd 100644 --- a/doc/pub/week34/html/._week34-bs037.html +++ b/doc/pub/week34/html/._week34-bs037.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    6254. Installing R, C++, cython or Julia
    6255. Installing R, C++, cython, Numba etc
    6256. Numpy examples and Important Matrix and vector handling packages
    6257. -
    6258. Basic Matrix Features
    6259. -
    6260.    Some famous Matrices
    6261. -
    6262.    More Basic Matrix Features
    6263. -
    6264. Numpy and arrays
    6265. -
    6266. Matrices in Python
    6267. -
    6268. Meet the Pandas
    6269. -
    6270. Friday August 27
    6271. -
    6272.    Simple linear regression model using scikit-learn
    6273. -
    6274.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6275. -
    6276.    Organizing our data
    6277. -
    6278.    Seeing the wood for the trees
    6279. -
    6280.    And what about using neural networks?
    6281. -
    6282. A first summary
    6283. -
    6284. Why Linear Regression (aka Ordinary Least Squares and family)
    6285. -
    6286. Regression analysis, overarching aims
    6287. -
    6288. Regression analysis, overarching aims II
    6289. -
    6290. Examples
    6291. -
    6292. General linear models
    6293. -
    6294. Rewriting the fitting procedure as a linear algebra problem
    6295. -
    6296. Rewriting the fitting procedure as a linear algebra problem, more details
    6297. -
    6298. Generalizing the fitting procedure as a linear algebra problem
    6299. -
    6300. Generalizing the fitting procedure as a linear algebra problem
    6301. -
    6302. Optimizing our parameters
    6303. -
    6304. Our model for the nuclear binding energies
    6305. -
    6306. Optimizing our parameters, more details
    6307. -
    6308. Interpretations and optimizing our parameters
    6309. -
    6310. Interpretations and optimizing our parameters
    6311. -
    6312. Some useful matrix and vector expressions
    6313. -
    6314. Interpretations and optimizing our parameters
    6315. -
    6316. Own code for Ordinary Least Squares
    6317. -
    6318. Adding error analysis and training set up
    6319. -
    6320. The \( \chi^2 \) function
    6321. -
    6322. The \( \chi^2 \) function
    6323. -
    6324. The \( \chi^2 \) function
    6325. -
    6326. The \( \chi^2 \) function
    6327. -
    6328. The \( \chi^2 \) function
    6329. -
    6330. The \( \chi^2 \) function
    6331. -
    6332. Fitting an Equation of State for Dense Nuclear Matter
    6333. -
    6334. The code
    6335. -
    6336. Splitting our Data in Training and Test data
    6337. -
    6338. Exercises
    6339. -
    6340. Exercise 1: Setting up various Python environments
    6341. -
    6342. Exercise 2: making your own data and exploring scikit-learn
    6343. -
    6344. Exercise 3: Normalizing our data
    6345. +
    6346. Numpy and arrays
    6347. +
    6348. Matrices in Python
    6349. +
    6350. Meet the Pandas
    6351. +
    6352.    Simple linear regression model using scikit-learn
    6353. +
    6354.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6355. +
    6356.    Organizing our data
    6357. +
    6358.    And what about using neural networks?
    6359. +
    6360. A first summary
    6361. +
    6362. Why Linear Regression (aka Ordinary Least Squares and family)
    6363. +
    6364. Regression analysis, overarching aims
    6365. +
    6366. Regression analysis, overarching aims II
    6367. +
    6368. Examples
    6369. +
    6370. General linear models
    6371. +
    6372. Rewriting the fitting procedure as a linear algebra problem
    6373. +
    6374. Rewriting the fitting procedure as a linear algebra problem, more details
    6375. +
    6376. Generalizing the fitting procedure as a linear algebra problem
    6377. +
    6378. Generalizing the fitting procedure as a linear algebra problem
    6379. +
    6380. Optimizing our parameters
    6381. +
    6382. Our model for the nuclear binding energies
    6383. +
    6384. Optimizing our parameters, more details
    6385. +
    6386. Interpretations and optimizing our parameters
    6387. +
    6388. Interpretations and optimizing our parameters
    6389. +
    6390. Some useful matrix and vector expressions
    6391. +
    6392. Interpretations and optimizing our parameters
    6393. +
    6394. Own code for Ordinary Least Squares
    6395. +
    6396. Adding error analysis and training set up
    6397. +
    6398. The \( \chi^2 \) function
    6399. +
    6400. The \( \chi^2 \) function
    6401. +
    6402. The \( \chi^2 \) function
    6403. +
    6404. The \( \chi^2 \) function
    6405. +
    6406. The \( \chi^2 \) function
    6407. +
    6408. The \( \chi^2 \) function
    6409. +
    6410. Fitting an Equation of State for Dense Nuclear Matter
    6411. +
    6412. The code
    6413. +
    6414. Splitting our Data in Training and Test data
    6415. +
    6416. Exercises
    6417. +
    6418. Exercise 1: Setting up various Python environments
    6419. +
    6420. Exercise 2: making your own data and exploring scikit-learn
    6421. +
    6422. Exercise 3: Split data in test and training data
    6423. @@ -351,914 +335,32 @@ MathJax.Hub.Config({

       

       

       

      -

      Simple linear regression model using scikit-learn

      - -

      We start with perhaps our simplest possible example, using Scikit-Learn to perform linear regression analysis on a data set produced by us.

      - -

      What follows is a simple Python code where we have defined a function -\( y \) in terms of the variable \( x \). Both are defined as vectors with \( 100 \) entries. -The numbers in the vector \( \boldsymbol{x} \) are given -by random numbers generated with a uniform distribution with entries -\( x_i \in [0,1] \) (more about probability distribution functions -later). These values are then used to define a function \( y(x) \) -(tabulated again as a vector) with a linear dependence on \( x \) plus a -random noise added via the normal distribution. +

      Examples

      +
      +
      + +

      In order to understand the relation among the predictors (or features or properties) \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), +consider the model we discussed for describing nuclear binding energies.

      -

      The Numpy functions are imported used the import numpy as np -statement and the random number generator for the uniform distribution -is called using the function np.random.rand(), where we specificy -that we want \( 100 \) random variables. Using Numpy we define -automatically an array with the specified number of elements, \( 100 \) in -our case. With the Numpy function randn() we can compute random -numbers with the normal distribution (mean value \( \mu \) equal to zero and -variance \( \sigma^2 \) set to one) and produce the values of \( y \) assuming a linear -dependence as function of \( x \) +

      There we assumed that we could parametrize the data using a polynomial approximation based on the liquid drop model. +Assuming

      - $$ -y = 2x+N(0,1), +BE(A) = a_0+a_1A+a_2A^{2/3}+a_3A^{-1/3}+a_4A^{-1}, $$ -

      where \( N(0,1) \) represents random numbers generated by the normal -distribution. From Scikit-Learn we import then the -LinearRegression functionality and make a prediction \( \tilde{y} = -\alpha + \beta x \) using the function fit(x,y). We call the set of -data \( (\boldsymbol{x},\boldsymbol{y}) \) for our training data. The Python package -scikit-learn has also a functionality which extracts the above -fitting parameters \( \alpha \) and \( \beta \) (see below). Later we will -distinguish between training data and test data. +

      we have five predictors, that is the intercept, the \( A \) dependent term, the \( A^{2/3} \) term and the \( A^{-1/3} \) and \( A^{-1} \) terms. +This gives \( p=0,1,2,3,4 \). Furthermore we have \( n \) entries for each predictor. It means that our design matrix is a +\( p\times n \) matrix \( \boldsymbol{X} \).

      -

      For plotting we use the Python package -matplotlib which produces publication -quality figures. Feel free to explore the extensive -gallery of examples. In -this example we plot our original values of \( x \) and \( y \) as well as the -prediction ypredict (\( \tilde{y} \)), which attempts at fitting our -data with a straight line. +

      Here the predictors are based on a model we have made. A popular data set which is widely encountered in ML applications is the +so-called credit card default data from Taiwan. The data set contains data on \( n=30000 \) credit card holders with predictors like gender, marital status, age, profession, education, etc. In total there are \( 24 \) such predictors or attributes leading to a design matrix of dimensionality \( 24 \times 30000 \). This is however a classification problem and we will come back to it when we discuss Logistic Regression.

      - -

      The Python code follows here.

      - - -
      -
      -
      -
      -
      -
      # Importing various packages
      -import numpy as np
      -import matplotlib.pyplot as plt
      -from sklearn.linear_model import LinearRegression
      -
      -x = np.random.rand(100,1)
      -y = 2*x+np.random.randn(100,1)
      -linreg = LinearRegression()
      -linreg.fit(x,y)
      -xnew = np.array([[0],[1]])
      -ypredict = linreg.predict(xnew)
      -
      -plt.plot(xnew, ypredict, "r-")
      -plt.plot(x, y ,'ro')
      -plt.axis([0,1.0,0, 5.0])
      -plt.xlabel(r'$x$')
      -plt.ylabel(r'$y$')
      -plt.title(r'Simple Linear Regression')
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      This example serves several aims. It allows us to demonstrate several -aspects of data analysis and later machine learning algorithms. The -immediate visualization shows that our linear fit is not -impressive. It goes through the data points, but there are many -outliers which are not reproduced by our linear regression. We could -now play around with this small program and change for example the -factor in front of \( x \) and the normal distribution. Try to change the -function \( y \) to -

      - -$$ -y = 10x+0.01 \times N(0,1), -$$ - -

      where \( x \) is defined as before. Does the fit look better? Indeed, by -reducing the role of the noise given by the normal distribution we see immediately that -our linear prediction seemingly reproduces better the training -set. However, this testing 'by the eye' is obviouly not satisfactory in the -long run. Here we have only defined the training data and our model, and -have not discussed a more rigorous approach to the cost function. -

      - -

      We need more rigorous criteria in defining whether we have succeeded or -not in modeling our training data. You will be surprised to see that -many scientists seldomly venture beyond this 'by the eye' approach. A -standard approach for the cost function is the so-called \( \chi^2 \) -function (a variant of the mean-squared error (MSE)) -

      - -$$ \chi^2 = \frac{1}{n} -\sum_{i=0}^{n-1}\frac{(y_i-\tilde{y}_i)^2}{\sigma_i^2}, -$$ - -

      where \( \sigma_i^2 \) is the variance (to be defined later) of the entry -\( y_i \). We may not know the explicit value of \( \sigma_i^2 \), it serves -however the aim of scaling the equations and make the cost function -dimensionless. -

      - -

      Minimizing the cost function is a central aspect of -our discussions to come. Finding its minima as function of the model -parameters (\( \alpha \) and \( \beta \) in our case) will be a recurring -theme in these series of lectures. Essentially all machine learning -algorithms we will discuss center around the minimization of the -chosen cost function. This depends in turn on our specific -model for describing the data, a typical situation in supervised -learning. Automatizing the search for the minima of the cost function is a -central ingredient in all algorithms. Typical methods which are -employed are various variants of gradient methods. These will be -discussed in more detail later. Again, you'll be surprised to hear that -many practitioners minimize the above function ''by the eye', popularly dubbed as -'chi by the eye'. That is, change a parameter and see (visually and numerically) that -the \( \chi^2 \) function becomes smaller. -

      - -

      There are many ways to define the cost function. A simpler approach is to look at the relative difference between the training data and the predicted data, that is we define -the relative error (why would we prefer the MSE instead of the relative error?) as -

      - -$$ -\epsilon_{\mathrm{relative}}= \frac{\vert \boldsymbol{y} -\boldsymbol{\tilde{y}}\vert}{\vert \boldsymbol{y}\vert}. -$$ - -

      The squared cost function results in an arithmetic mean-unbiased -estimator, and the absolute-value cost function results in a -median-unbiased estimator (in the one-dimensional case, and a -geometric median-unbiased estimator for the multi-dimensional -case). The squared cost function has the disadvantage that it has the tendency -to be dominated by outliers. -

      - -

      We can modify easily the above Python code and plot the relative error instead

      - - -
      -
      -
      -
      -
      -
      import numpy as np
      -import matplotlib.pyplot as plt
      -from sklearn.linear_model import LinearRegression
      -
      -x = np.random.rand(100,1)
      -y = 5*x+0.01*np.random.randn(100,1)
      -linreg = LinearRegression()
      -linreg.fit(x,y)
      -ypredict = linreg.predict(x)
      -
      -plt.plot(x, np.abs(ypredict-y)/abs(y), "ro")
      -plt.axis([0,1.0,0.0, 0.5])
      -plt.xlabel(r'$x$')
      -plt.ylabel(r'$\epsilon_{\mathrm{relative}}$')
      -plt.title(r'Relative error')
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Depending on the parameter in front of the normal distribution, we may -have a small or larger relative error. Try to play around with -different training data sets and study (graphically) the value of the -relative error. -

      - -

      As mentioned above, Scikit-Learn has an impressive functionality. -We can for example extract the values of \( \alpha \) and \( \beta \) and -their error estimates, or the variance and standard deviation and many -other properties from the statistical data analysis. -

      - -

      Here we show an -example of the functionality of Scikit-Learn. -

      - - -
      -
      -
      -
      -
      -
      import numpy as np 
      -import matplotlib.pyplot as plt 
      -from sklearn.linear_model import LinearRegression 
      -from sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error, mean_absolute_error
      -
      -x = np.random.rand(100,1)
      -y = 2.0+ 5*x+0.5*np.random.randn(100,1)
      -linreg = LinearRegression()
      -linreg.fit(x,y)
      -ypredict = linreg.predict(x)
      -print('The intercept alpha: \n', linreg.intercept_)
      -print('Coefficient beta : \n', linreg.coef_)
      -# The mean squared error                               
      -print("Mean squared error: %.2f" % mean_squared_error(y, ypredict))
      -# Explained variance score: 1 is perfect prediction                                 
      -print('Variance score: %.2f' % r2_score(y, ypredict))
      -# Mean squared log error                                                        
      -print('Mean squared log error: %.2f' % mean_squared_log_error(y, ypredict) )
      -# Mean absolute error                                                           
      -print('Mean absolute error: %.2f' % mean_absolute_error(y, ypredict))
      -plt.plot(x, ypredict, "r-")
      -plt.plot(x, y ,'ro')
      -plt.axis([0.0,1.0,1.5, 7.0])
      -plt.xlabel(r'$x$')
      -plt.ylabel(r'$y$')
      -plt.title(r'Linear Regression fit ')
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      The function coef gives us the parameter \( \beta \) of our fit while intercept yields -\( \alpha \). Depending on the constant in front of the normal distribution, we get values near or far from \( \alpha =2 \) and \( \beta =5 \). Try to play around with different parameters in front of the normal distribution. The function meansquarederror gives us the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error or loss defined as -

      -$$ MSE(\boldsymbol{y},\boldsymbol{\tilde{y}}) = \frac{1}{n} -\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, -$$ - -

      The smaller the value, the better the fit. Ideally we would like to -have an MSE equal zero. The attentive reader has probably recognized -this function as being similar to the \( \chi^2 \) function defined above. -

      - -

      The r2score function computes \( R^2 \), the coefficient of -determination. It provides a measure of how well future samples are -likely to be predicted by the model. Best possible score is 1.0 and it -can be negative (because the model can be arbitrarily worse). A -constant model that always predicts the expected value of \( \boldsymbol{y} \), -disregarding the input features, would get a \( R^2 \) score of \( 0.0 \). -

      - -

      If \( \tilde{\boldsymbol{y}}_i \) is the predicted value of the \( i-th \) sample and \( y_i \) is the corresponding true value, then the score \( R^2 \) is defined as

      -$$ -R^2(\boldsymbol{y}, \tilde{\boldsymbol{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, -$$ - -

      where we have defined the mean value of \( \boldsymbol{y} \) as

      -$$ -\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. -$$ - -

      Another quantity taht we will meet again in our discussions of regression analysis is - the mean absolute error (MAE), a risk metric corresponding to the expected value of the absolute error loss or what we call the \( l1 \)-norm loss. In our discussion above we presented the relative error. -The MAE is defined as follows -

      -$$ -\text{MAE}(\boldsymbol{y}, \boldsymbol{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n-1} \left| y_i - \tilde{y}_i \right|. -$$ - -

      We present the -squared logarithmic (quadratic) error -

      -$$ -\text{MSLE}(\boldsymbol{y}, \boldsymbol{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n - 1} (\log_e (1 + y_i) - \log_e (1 + \tilde{y}_i) )^2, -$$ - -

      where \( \log_e (x) \) stands for the natural logarithm of \( x \). This error -estimate is best to use when targets having exponential growth, such -as population counts, average sales of a commodity over a span of -years etc. -

      - -

      Finally, another cost function is the Huber cost function used in robust regression.

      - -

      The rationale behind this possible cost function is its reduced -sensitivity to outliers in the data set. In our discussions on -dimensionality reduction and normalization of data we will meet other -ways of dealing with outliers. -

      - -

      The Huber cost function is defined as

      -$$ -H_{\delta}(\boldsymbol{a})=\left\{\begin{array}{cc}\frac{1}{2} \boldsymbol{a}^{2}& \text{for }|\boldsymbol{a}|\leq \delta\\ \delta (|\boldsymbol{a}|-\frac{1}{2}\delta ),&\text{otherwise}.\end{array}\right. -$$ - -

      Here \( \boldsymbol{a}=\boldsymbol{y} - \boldsymbol{\tilde{y}} \).

      - -

      We will discuss in more detail these and other functions in the -various lectures. We conclude this part with another example. Instead -of a linear \( x \)-dependence we study now a cubic polynomial and use the -polynomial regression analysis tools of scikit-learn. -

      - - - -
      -
      -
      -
      -
      -
      import matplotlib.pyplot as plt
      -import numpy as np
      -import random
      -from sklearn.linear_model import Ridge
      -from sklearn.preprocessing import PolynomialFeatures
      -from sklearn.pipeline import make_pipeline
      -from sklearn.linear_model import LinearRegression
      -
      -x=np.linspace(0.02,0.98,200)
      -noise = np.asarray(random.sample((range(200)),200))
      -y=x**3*noise
      -yn=x**3*100
      -poly3 = PolynomialFeatures(degree=3)
      -X = poly3.fit_transform(x[:,np.newaxis])
      -clf3 = LinearRegression()
      -clf3.fit(X,y)
      -
      -Xplot=poly3.fit_transform(x[:,np.newaxis])
      -poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')
      -plt.plot(x,yn, color='red', label="True Cubic")
      -plt.scatter(x, y, label='Data', color='orange', s=15)
      -plt.legend()
      -plt.show()
      -
      -def error(a):
      -    for i in y:
      -        err=(y-yn)/yn
      -    return abs(np.sum(err))/len(err)
      -
      -print (error(y))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      To our real data: nuclear binding energies. Brief reminder on masses and binding energies

      - -

      Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding -energies. A basic quantity which can be measured for the ground -states of nuclei is the atomic mass \( M(N, Z) \) of the neutral atom with -atomic mass number \( A \) and charge \( Z \). The number of neutrons is \( N \). There are indeed several sophisticated experiments worldwide which allow us to measure this quantity to high precision (parts per million even). -

      - -

      Atomic masses are usually tabulated in terms of the mass excess defined by

      -$$ -\Delta M(N, Z) = M(N, Z) - uA, -$$ - -

      where \( u \) is the Atomic Mass Unit

      -$$ -u = M(^{12}\mathrm{C})/12 = 931.4940954(57) \hspace{0.1cm} \mathrm{MeV}/c^2. -$$ - -

      The nucleon masses are

      -$$ -m_p = 1.00727646693(9)u, -$$ - -

      and

      -$$ -m_n = 939.56536(8)\hspace{0.1cm} \mathrm{MeV}/c^2 = 1.0086649156(6)u. -$$ - -

      In the 2016 mass evaluation of by W.J.Huang, G.Audi, M.Wang, F.G.Kondev, S.Naimi and X.Xu -there are data on masses and decays of 3437 nuclei. -

      - -

      The nuclear binding energy is defined as the energy required to break -up a given nucleus into its constituent parts of \( N \) neutrons and \( Z \) -protons. In terms of the atomic masses \( M(N, Z) \) the binding energy is -defined by -

      - -$$ -BE(N, Z) = ZM_H c^2 + Nm_n c^2 - M(N, Z)c^2 , -$$ - -

      where \( M_H \) is the mass of the hydrogen atom and \( m_n \) is the mass of the neutron. -In terms of the mass excess the binding energy is given by -

      -$$ -BE(N, Z) = Z\Delta_H c^2 + N\Delta_n c^2 -\Delta(N, Z)c^2 , -$$ - -

      where \( \Delta_H c^2 = 7.2890 \) MeV and \( \Delta_n c^2 = 8.0713 \) MeV.

      - -

      A popular and physically intuitive model which can be used to parametrize -the experimental binding energies as function of \( A \), is the so-called -liquid drop model. The ansatz is based on the following expression -

      - -$$ -BE(N,Z) = a_1A-a_2A^{2/3}-a_3\frac{Z^2}{A^{1/3}}-a_4\frac{(N-Z)^2}{A}, -$$ - -

      where \( A \) stands for the number of nucleons and the $a_i$s are parameters which are determined by a fit -to the experimental data. -

      - -

      To arrive at the above expression we have assumed that we can make the following assumptions:

      - -
        -
      • There is a volume term \( a_1A \) proportional with the number of nucleons (the energy is also an extensive quantity). When an assembly of nucleons of the same size is packed together into the smallest volume, each interior nucleon has a certain number of other nucleons in contact with it. This contribution is proportional to the volume.
      • -
      • There is a surface energy term \( a_2A^{2/3} \). The assumption here is that a nucleon at the surface of a nucleus interacts with fewer other nucleons than one in the interior of the nucleus and hence its binding energy is less. This surface energy term takes that into account and is therefore negative and is proportional to the surface area.
      • -
      • There is a Coulomb energy term \( a_3\frac{Z^2}{A^{1/3}} \). The electric repulsion between each pair of protons in a nucleus yields less binding.
      • -
      • There is an asymmetry term \( a_4\frac{(N-Z)^2}{A} \). This term is associated with the Pauli exclusion principle and reflects the fact that the proton-neutron interaction is more attractive on the average than the neutron-neutron and proton-proton interactions.
      • -
      -

      We could also add a so-called pairing term, which is a correction term that -arises from the tendency of proton pairs and neutron pairs to -occur. An even number of particles is more stable than an odd number. -

      -

      Organizing our data

      - -

      Let us start with reading and organizing our data. -We start with the compilation of masses and binding energies from 2016. -After having downloaded this file to our own computer, we are now ready to read the file and start structuring our data. -

      - -

      We start with preparing folders for storing our calculations and the data file over masses and binding energies. We import also various modules that we will find useful in order to present various Machine Learning methods. Here we focus mainly on the functionality of scikit-learn.

      - - -
      -
      -
      -
      -
      -
      # Common imports
      -import numpy as np
      -import pandas as pd
      -import matplotlib.pyplot as plt
      -import sklearn.linear_model as skl
      -from sklearn.model_selection import train_test_split
      -from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
      -import os
      -
      -# Where to save the figures and data files
      -PROJECT_ROOT_DIR = "Results"
      -FIGURE_ID = "Results/FigureFiles"
      -DATA_ID = "DataFiles/"
      -
      -if not os.path.exists(PROJECT_ROOT_DIR):
      -    os.mkdir(PROJECT_ROOT_DIR)
      -
      -if not os.path.exists(FIGURE_ID):
      -    os.makedirs(FIGURE_ID)
      -
      -if not os.path.exists(DATA_ID):
      -    os.makedirs(DATA_ID)
      -
      -def image_path(fig_id):
      -    return os.path.join(FIGURE_ID, fig_id)
      -
      -def data_path(dat_id):
      -    return os.path.join(DATA_ID, dat_id)
      -
      -def save_fig(fig_id):
      -    plt.savefig(image_path(fig_id) + ".png", format='png')
      -
      -infile = open(data_path("MassEval2016.dat"),'r')
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Before we proceed, we define also a function for making our plots. You can obviously avoid this and simply set up various matplotlib commands every time you need them. You may however find it convenient to collect all such commands in one function and simply call this function.

      - - -
      -
      -
      -
      -
      -
      from pylab import plt, mpl
      -plt.style.use('seaborn')
      -mpl.rcParams['font.family'] = 'serif'
      -
      -def MakePlot(x,y, styles, labels, axlabels):
      -    plt.figure(figsize=(10,6))
      -    for i in range(len(x)):
      -        plt.plot(x[i], y[i], styles[i], label = labels[i])
      -        plt.xlabel(axlabels[0])
      -        plt.ylabel(axlabels[1])
      -    plt.legend(loc=0)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Our next step is to read the data on experimental binding energies and -reorganize them as functions of the mass number \( A \), the number of -protons \( Z \) and neutrons \( N \) using pandas. Before we do this it is -always useful (unless you have a binary file or other types of compressed -data) to actually open the file and simply take a look at it! -

      - -

      In particular, the program that outputs the final nuclear masses is written in Fortran with a specific format. It means that we need to figure out the format and which columns contain the data we are interested in. Pandas comes with a function that reads formatted output. After having admired the file, we are now ready to start massaging it with pandas. The file begins with some basic format information.

      - - -
      -
      -
      -
      -
      -
      """                                                                                                                         
      -This is taken from the data file of the mass 2016 evaluation.                                                               
      -All files are 3436 lines long with 124 character per line.                                                                  
      -       Headers are 39 lines long.                                                                                           
      -   col 1     :  Fortran character control: 1 = page feed  0 = line feed                                                     
      -   format    :  a1,i3,i5,i5,i5,1x,a3,a4,1x,f13.5,f11.5,f11.3,f9.3,1x,a2,f11.3,f9.3,1x,i3,1x,f12.5,f11.5                     
      -   These formats are reflected in the pandas widths variable below, see the statement                                       
      -   widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),                                                            
      -   Pandas has also a variable header, with length 39 in this case.                                                          
      -"""
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      The data we are interested in are in columns 2, 3, 4 and 11, giving us -the number of neutrons, protons, mass numbers and binding energies, -respectively. We add also for the sake of completeness the element name. The data are in fixed-width formatted lines and we will -covert them into the pandas DataFrame structure. -

      - - - -
      -
      -
      -
      -
      -
      # Read the experimental data with Pandas
      -Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),
      -              names=('N', 'Z', 'A', 'Element', 'Ebinding'),
      -              widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),
      -              header=39,
      -              index_col=False)
      -
      -# Extrapolated values are indicated by '#' in place of the decimal place, so
      -# the Ebinding column won't be numeric. Coerce to float and drop these entries.
      -Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')
      -Masses = Masses.dropna()
      -# Convert from keV to MeV.
      -Masses['Ebinding'] /= 1000
      -
      -# Group the DataFrame by nucleon number, A.
      -Masses = Masses.groupby('A')
      -# Find the rows of the grouped DataFrame with the maximum binding energy.
      -Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      We have now read in the data, grouped them according to the variables we are interested in. -We see how easy it is to reorganize the data using pandas. If we -were to do these operations in C/C++ or Fortran, we would have had to -write various functions/subroutines which perform the above -reorganizations for us. Having reorganized the data, we can now start -to make some simple fits using both the functionalities in numpy and -Scikit-Learn afterwards. -

      - -

      Now we define five variables which contain -the number of nucleons \( A \), the number of protons \( Z \) and the number of neutrons \( N \), the element name and finally the energies themselves. -

      - - -
      -
      -
      -
      -
      -
      A = Masses['A']
      -Z = Masses['Z']
      -N = Masses['N']
      -Element = Masses['Element']
      -Energies = Masses['Ebinding']
      -print(Masses)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      The next step, and we will define this mathematically later, is to set up the so-called design matrix. We will throughout call this matrix \( \boldsymbol{X} \). -It has dimensionality \( p\times n \), where \( n \) is the number of data points and \( p \) are the so-called predictors. In our case here they are given by the number of polynomials in \( A \) we wish to include in the fit. -

      - - -
      -
      -
      -
      -
      -
      # Now we set up the design matrix X
      -X = np.zeros((len(A),5))
      -X[:,0] = 1
      -X[:,1] = A
      -X[:,2] = A**(2.0/3.0)
      -X[:,3] = A**(-1.0/3.0)
      -X[:,4] = A**(-1.0)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      With scikitlearn we are now ready to use linear regression and fit our data.

      - - -
      -
      -
      -
      -
      -
      clf = skl.LinearRegression().fit(X, Energies)
      -fity = clf.predict(X)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Pretty simple! -Now we can print measures of how our fit is doing, the coefficients from the fits and plot the final fit together with our data. -

      - - -
      -
      -
      -
      -
      -
      # The mean squared error                               
      -print("Mean squared error: %.2f" % mean_squared_error(Energies, fity))
      -# Explained variance score: 1 is perfect prediction                                 
      -print('Variance score: %.2f' % r2_score(Energies, fity))
      -# Mean absolute error                                                           
      -print('Mean absolute error: %.2f' % mean_absolute_error(Energies, fity))
      -print(clf.coef_, clf.intercept_)
      -
      -Masses['Eapprox']  = fity
      -# Generate a plot comparing the experimental with the fitted values values.
      -fig, ax = plt.subplots()
      -ax.set_xlabel(r'$A = N + Z$')
      -ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$')
      -ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,
      -            label='Ame2016')
      -ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',
      -            label='Fit')
      -ax.legend()
      -save_fig("Masses2016")
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      Seeing the wood for the trees

      - -

      As a teaser, let us now see how we can do this with decision trees using scikit-learn. Later we will switch to so-called random forests!

      - - - -
      -
      -
      -
      -
      -
      #Decision Tree Regression
      -from sklearn.tree import DecisionTreeRegressor
      -regr_1=DecisionTreeRegressor(max_depth=5)
      -regr_2=DecisionTreeRegressor(max_depth=7)
      -regr_3=DecisionTreeRegressor(max_depth=9)
      -regr_1.fit(X, Energies)
      -regr_2.fit(X, Energies)
      -regr_3.fit(X, Energies)
      -
      -
      -y_1 = regr_1.predict(X)
      -y_2 = regr_2.predict(X)
      -y_3=regr_3.predict(X)
      -Masses['Eapprox'] = y_3
      -# Plot the results
      -plt.figure()
      -plt.plot(A, Energies, color="blue", label="Data", linewidth=2)
      -plt.plot(A, y_1, color="red", label="max_depth=5", linewidth=2)
      -plt.plot(A, y_2, color="green", label="max_depth=7", linewidth=2)
      -plt.plot(A, y_3, color="m", label="max_depth=9", linewidth=2)
      -
      -plt.xlabel("$A$")
      -plt.ylabel("$E$[MeV]")
      -plt.title("Decision Tree Regression")
      -plt.legend()
      -save_fig("Masses2016Trees")
      -plt.show()
      -print(Masses)
      -print(np.mean( (Energies-y_1)**2))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      And what about using neural networks?

      -

      The seaborn package allows us to visualize data in an efficient way. Note that we use scikit-learn's multi-layer perceptron (or feed forward neural network) -functionality. -

      - - -
      -
      -
      -
      -
      -
      from sklearn.neural_network import MLPRegressor
      -from sklearn.metrics import accuracy_score
      -import seaborn as sns
      -
      -X_train = X
      -Y_train = Energies
      -n_hidden_neurons = 100
      -epochs = 100
      -# store models for later use
      -eta_vals = np.logspace(-5, 1, 7)
      -lmbd_vals = np.logspace(-5, 1, 7)
      -# store the models for later use
      -DNN_scikit = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
      -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
      -sns.set()
      -for i, eta in enumerate(eta_vals):
      -    for j, lmbd in enumerate(lmbd_vals):
      -        dnn = MLPRegressor(hidden_layer_sizes=(n_hidden_neurons), activation='logistic',
      -                            alpha=lmbd, learning_rate_init=eta, max_iter=epochs)
      -        dnn.fit(X_train, Y_train)
      -        DNN_scikit[i][j] = dnn
      -        train_accuracy[i][j] = dnn.score(X_train, Y_train)
      -
      -fig, ax = plt.subplots(figsize = (10, 10))
      -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
      -ax.set_title("Training Accuracy")
      -ax.set_ylabel("$\eta$")
      -ax.set_xlabel("$\lambda$")
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      A first summary

      - -

      The aim behind these introductory words was to present to you various -Python libraries and their functionalities, in particular libraries like -numpy, pandas, xarray and matplotlib and other that make our life much easier -in handling various data sets and visualizing data. -

      - -

      Furthermore, -Scikit-Learn allows us with few lines of code to implement popular -Machine Learning algorithms for supervised learning. Later we will meet Tensorflow, a powerful library for deep learning. -Now it is time to dive more into the details of various methods. We will start with linear regression and try to take a deeper look at what it entails. -

      @@ -1285,7 +387,7 @@ Now it is time to dive more into the details of various methods. We will start w

    6424. 46
    6425. 47
    6426. ...
    6427. -
    6428. 66
    6429. +
    6430. 62
    6431. »
    6432. diff --git a/doc/pub/week34/html/._week34-bs038.html b/doc/pub/week34/html/._week34-bs038.html index f3b259e59..1c16ab3e8 100644 --- a/doc/pub/week34/html/._week34-bs038.html +++ b/doc/pub/week34/html/._week34-bs038.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    6433. Installing R, C++, cython or Julia
    6434. Installing R, C++, cython, Numba etc
    6435. Numpy examples and Important Matrix and vector handling packages
    6436. -
    6437. Basic Matrix Features
    6438. -
    6439.    Some famous Matrices
    6440. -
    6441.    More Basic Matrix Features
    6442. -
    6443. Numpy and arrays
    6444. -
    6445. Matrices in Python
    6446. -
    6447. Meet the Pandas
    6448. -
    6449. Friday August 27
    6450. -
    6451.    Simple linear regression model using scikit-learn
    6452. -
    6453.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6454. -
    6455.    Organizing our data
    6456. -
    6457.    Seeing the wood for the trees
    6458. -
    6459.    And what about using neural networks?
    6460. -
    6461. A first summary
    6462. -
    6463. Why Linear Regression (aka Ordinary Least Squares and family)
    6464. -
    6465. Regression analysis, overarching aims
    6466. -
    6467. Regression analysis, overarching aims II
    6468. -
    6469. Examples
    6470. -
    6471. General linear models
    6472. -
    6473. Rewriting the fitting procedure as a linear algebra problem
    6474. -
    6475. Rewriting the fitting procedure as a linear algebra problem, more details
    6476. -
    6477. Generalizing the fitting procedure as a linear algebra problem
    6478. -
    6479. Generalizing the fitting procedure as a linear algebra problem
    6480. -
    6481. Optimizing our parameters
    6482. -
    6483. Our model for the nuclear binding energies
    6484. -
    6485. Optimizing our parameters, more details
    6486. -
    6487. Interpretations and optimizing our parameters
    6488. -
    6489. Interpretations and optimizing our parameters
    6490. -
    6491. Some useful matrix and vector expressions
    6492. -
    6493. Interpretations and optimizing our parameters
    6494. -
    6495. Own code for Ordinary Least Squares
    6496. -
    6497. Adding error analysis and training set up
    6498. -
    6499. The \( \chi^2 \) function
    6500. -
    6501. The \( \chi^2 \) function
    6502. -
    6503. The \( \chi^2 \) function
    6504. -
    6505. The \( \chi^2 \) function
    6506. -
    6507. The \( \chi^2 \) function
    6508. -
    6509. The \( \chi^2 \) function
    6510. -
    6511. Fitting an Equation of State for Dense Nuclear Matter
    6512. -
    6513. The code
    6514. -
    6515. Splitting our Data in Training and Test data
    6516. -
    6517. Exercises
    6518. -
    6519. Exercise 1: Setting up various Python environments
    6520. -
    6521. Exercise 2: making your own data and exploring scikit-learn
    6522. -
    6523. Exercise 3: Normalizing our data
    6524. +
    6525. Numpy and arrays
    6526. +
    6527. Matrices in Python
    6528. +
    6529. Meet the Pandas
    6530. +
    6531.    Simple linear regression model using scikit-learn
    6532. +
    6533.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6534. +
    6535.    Organizing our data
    6536. +
    6537.    And what about using neural networks?
    6538. +
    6539. A first summary
    6540. +
    6541. Why Linear Regression (aka Ordinary Least Squares and family)
    6542. +
    6543. Regression analysis, overarching aims
    6544. +
    6545. Regression analysis, overarching aims II
    6546. +
    6547. Examples
    6548. +
    6549. General linear models
    6550. +
    6551. Rewriting the fitting procedure as a linear algebra problem
    6552. +
    6553. Rewriting the fitting procedure as a linear algebra problem, more details
    6554. +
    6555. Generalizing the fitting procedure as a linear algebra problem
    6556. +
    6557. Generalizing the fitting procedure as a linear algebra problem
    6558. +
    6559. Optimizing our parameters
    6560. +
    6561. Our model for the nuclear binding energies
    6562. +
    6563. Optimizing our parameters, more details
    6564. +
    6565. Interpretations and optimizing our parameters
    6566. +
    6567. Interpretations and optimizing our parameters
    6568. +
    6569. Some useful matrix and vector expressions
    6570. +
    6571. Interpretations and optimizing our parameters
    6572. +
    6573. Own code for Ordinary Least Squares
    6574. +
    6575. Adding error analysis and training set up
    6576. +
    6577. The \( \chi^2 \) function
    6578. +
    6579. The \( \chi^2 \) function
    6580. +
    6581. The \( \chi^2 \) function
    6582. +
    6583. The \( \chi^2 \) function
    6584. +
    6585. The \( \chi^2 \) function
    6586. +
    6587. The \( \chi^2 \) function
    6588. +
    6589. Fitting an Equation of State for Dense Nuclear Matter
    6590. +
    6591. The code
    6592. +
    6593. Splitting our Data in Training and Test data
    6594. +
    6595. Exercises
    6596. +
    6597. Exercise 1: Setting up various Python environments
    6598. +
    6599. Exercise 2: making your own data and exploring scikit-learn
    6600. +
    6601. Exercise 3: Split data in test and training data
    6602. @@ -351,23 +335,21 @@ MathJax.Hub.Config({

       

       

       

      -

      Why Linear Regression (aka Ordinary Least Squares and family)

      +

      General linear models

      +
      +
      + +

      Before we proceed let us study a case from linear algebra where we aim at fitting a set of data \( \boldsymbol{y}=[y_0,y_1,\dots,y_{n-1}] \). We could think of these data as a result of an experiment or a complicated numerical experiment. These data are functions of a series of variables \( \boldsymbol{x}=[x_0,x_1,\dots,x_{n-1}] \), that is \( y_i = y(x_i) \) with \( i=0,1,2,\dots,n-1 \). The variables \( x_i \) could represent physical quantities like time, temperature, position etc. We assume that \( y(x) \) is a smooth function.

      + +

      Since obtaining these data points may not be trivial, we want to use these data to fit a function which can allow us to make predictions for values of \( y \) which are not in the present set. The perhaps simplest approach is to assume we can parametrize our function in terms of a polynomial of degree \( n-1 \) with \( n \) points, that is

      +$$ +y=y(x) \rightarrow y(x_i)=\tilde{y}_i+\epsilon_i=\sum_{j=0}^{n-1} \beta_j x_i^j+\epsilon_i, +$$ + +

      where \( \epsilon_i \) is the error in our approximation.

      +
      +
      -

      Fitting a continuous function with linear parameterization in terms of the parameters \( \boldsymbol{\beta} \).

      -
        -
      • Method of choice for fitting a continuous function!
      • -
      • Gives an excellent introduction to central Machine Learning features with understandable pedagogical links to other methods like Neural Networks, Support Vector Machines etc
      • -
      • Analytical expression for the fitting parameters \( \boldsymbol{\beta} \)
      • -
      • Analytical expressions for statistical propertiers like mean values, variances, confidence intervals and more
      • -
      • Analytical relation with probabilistic interpretations
      • -
      • Easy to introduce basic concepts like bias-variance tradeoff, cross-validation, resampling and regularization techniques and many other ML topics
      • -
      • Easy to code! And links well with classification problems and logistic regression and neural networks
      • -
      • Allows for easy hands-on understanding of gradient descent methods
      • -
      • and many more features
      • -
      -

      For more discussions of Ridge and Lasso regression, Wessel van Wieringen's article is highly recommended. -Similarly, Mehta et al's article is also recommended. -

      @@ -394,7 +376,7 @@ Similarly, Mehta et al

    6603. 47
    6604. 48
    6605. ...
    6606. -
    6607. 66
    6608. +
    6609. 62
    6610. »
    6611. diff --git a/doc/pub/week34/html/._week34-bs039.html b/doc/pub/week34/html/._week34-bs039.html index b4a839241..17aa7bf6e 100644 --- a/doc/pub/week34/html/._week34-bs039.html +++ b/doc/pub/week34/html/._week34-bs039.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    6612. Installing R, C++, cython or Julia
    6613. Installing R, C++, cython, Numba etc
    6614. Numpy examples and Important Matrix and vector handling packages
    6615. -
    6616. Basic Matrix Features
    6617. -
    6618.    Some famous Matrices
    6619. -
    6620.    More Basic Matrix Features
    6621. -
    6622. Numpy and arrays
    6623. -
    6624. Matrices in Python
    6625. -
    6626. Meet the Pandas
    6627. -
    6628. Friday August 27
    6629. -
    6630.    Simple linear regression model using scikit-learn
    6631. -
    6632.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6633. -
    6634.    Organizing our data
    6635. -
    6636.    Seeing the wood for the trees
    6637. -
    6638.    And what about using neural networks?
    6639. -
    6640. A first summary
    6641. -
    6642. Why Linear Regression (aka Ordinary Least Squares and family)
    6643. -
    6644. Regression analysis, overarching aims
    6645. -
    6646. Regression analysis, overarching aims II
    6647. -
    6648. Examples
    6649. -
    6650. General linear models
    6651. -
    6652. Rewriting the fitting procedure as a linear algebra problem
    6653. -
    6654. Rewriting the fitting procedure as a linear algebra problem, more details
    6655. -
    6656. Generalizing the fitting procedure as a linear algebra problem
    6657. -
    6658. Generalizing the fitting procedure as a linear algebra problem
    6659. -
    6660. Optimizing our parameters
    6661. -
    6662. Our model for the nuclear binding energies
    6663. -
    6664. Optimizing our parameters, more details
    6665. -
    6666. Interpretations and optimizing our parameters
    6667. -
    6668. Interpretations and optimizing our parameters
    6669. -
    6670. Some useful matrix and vector expressions
    6671. -
    6672. Interpretations and optimizing our parameters
    6673. -
    6674. Own code for Ordinary Least Squares
    6675. -
    6676. Adding error analysis and training set up
    6677. -
    6678. The \( \chi^2 \) function
    6679. -
    6680. The \( \chi^2 \) function
    6681. -
    6682. The \( \chi^2 \) function
    6683. -
    6684. The \( \chi^2 \) function
    6685. -
    6686. The \( \chi^2 \) function
    6687. -
    6688. The \( \chi^2 \) function
    6689. -
    6690. Fitting an Equation of State for Dense Nuclear Matter
    6691. -
    6692. The code
    6693. -
    6694. Splitting our Data in Training and Test data
    6695. -
    6696. Exercises
    6697. -
    6698. Exercise 1: Setting up various Python environments
    6699. -
    6700. Exercise 2: making your own data and exploring scikit-learn
    6701. -
    6702. Exercise 3: Normalizing our data
    6703. +
    6704. Numpy and arrays
    6705. +
    6706. Matrices in Python
    6707. +
    6708. Meet the Pandas
    6709. +
    6710.    Simple linear regression model using scikit-learn
    6711. +
    6712.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6713. +
    6714.    Organizing our data
    6715. +
    6716.    And what about using neural networks?
    6717. +
    6718. A first summary
    6719. +
    6720. Why Linear Regression (aka Ordinary Least Squares and family)
    6721. +
    6722. Regression analysis, overarching aims
    6723. +
    6724. Regression analysis, overarching aims II
    6725. +
    6726. Examples
    6727. +
    6728. General linear models
    6729. +
    6730. Rewriting the fitting procedure as a linear algebra problem
    6731. +
    6732. Rewriting the fitting procedure as a linear algebra problem, more details
    6733. +
    6734. Generalizing the fitting procedure as a linear algebra problem
    6735. +
    6736. Generalizing the fitting procedure as a linear algebra problem
    6737. +
    6738. Optimizing our parameters
    6739. +
    6740. Our model for the nuclear binding energies
    6741. +
    6742. Optimizing our parameters, more details
    6743. +
    6744. Interpretations and optimizing our parameters
    6745. +
    6746. Interpretations and optimizing our parameters
    6747. +
    6748. Some useful matrix and vector expressions
    6749. +
    6750. Interpretations and optimizing our parameters
    6751. +
    6752. Own code for Ordinary Least Squares
    6753. +
    6754. Adding error analysis and training set up
    6755. +
    6756. The \( \chi^2 \) function
    6757. +
    6758. The \( \chi^2 \) function
    6759. +
    6760. The \( \chi^2 \) function
    6761. +
    6762. The \( \chi^2 \) function
    6763. +
    6764. The \( \chi^2 \) function
    6765. +
    6766. The \( \chi^2 \) function
    6767. +
    6768. Fitting an Equation of State for Dense Nuclear Matter
    6769. +
    6770. The code
    6771. +
    6772. Splitting our Data in Training and Test data
    6773. +
    6774. Exercises
    6775. +
    6776. Exercise 1: Setting up various Python environments
    6777. +
    6778. Exercise 2: making your own data and exploring scikit-learn
    6779. +
    6780. Exercise 3: Split data in test and training data
    6781. @@ -351,22 +335,20 @@ MathJax.Hub.Config({

       

       

       

      -

      Regression analysis, overarching aims

      +

      Rewriting the fitting procedure as a linear algebra problem

      - -

      Regression modeling deals with the description of the sampling distribution of a given random variable \( y \) and how it varies as function of another variable or a set of such variables \( \boldsymbol{x} =[x_0, x_1,\dots, x_{n-1}]^T \). -The first variable is called the dependent, the outcome or the response variable while the set of variables \( \boldsymbol{x} \) is called the independent variable, or the predictor variable or the explanatory variable. -

      - -

      A regression model aims at finding a likelihood function \( p(\boldsymbol{y}\vert \boldsymbol{x}) \), that is the conditional distribution for \( \boldsymbol{y} \) with a given \( \boldsymbol{x} \). The estimation of \( p(\boldsymbol{y}\vert \boldsymbol{x}) \) is made using a data set with

      -
        -
      • \( n \) cases \( i = 0, 1, 2, \dots, n-1 \)
      • -
      • Response (target, dependent or outcome) variable \( y_i \) with \( i = 0, 1, 2, \dots, n-1 \)
      • -
      • \( p \) so-called explanatory (independent or predictor) variables \( \boldsymbol{x}_i=[x_{i0}, x_{i1}, \dots, x_{ip-1}] \) with \( i = 0, 1, 2, \dots, n-1 \) and explanatory variables running from \( 0 \) to \( p-1 \). See below for more explicit examples.
      • -
      -

      The goal of the regression analysis is to extract/exploit relationship between \( \boldsymbol{y} \) and \( \boldsymbol{x} \) in or to infer causal dependencies, approximations to the likelihood functions, functional relationships and to make predictions, making fits and many other things.

      +

      For every set of values \( y_i,x_i \) we have thus the corresponding set of equations

      +$$ +\begin{align*} +y_0&=\beta_0+\beta_1x_0^1+\beta_2x_0^2+\dots+\beta_{n-1}x_0^{n-1}+\epsilon_0\\ +y_1&=\beta_0+\beta_1x_1^1+\beta_2x_1^2+\dots+\beta_{n-1}x_1^{n-1}+\epsilon_1\\ +y_2&=\beta_0+\beta_1x_2^1+\beta_2x_2^2+\dots+\beta_{n-1}x_2^{n-1}+\epsilon_2\\ +\dots & \dots \\ +y_{n-1}&=\beta_0+\beta_1x_{n-1}^1+\beta_2x_{n-1}^2+\dots+\beta_{n-1}x_{n-1}^{n-1}+\epsilon_{n-1}.\\ +\end{align*} +$$
      @@ -396,7 +378,7 @@ The first variable is called the dependent, the outcome or the
    6782. 48
    6783. 49
    6784. ...
    6785. -
    6786. 66
    6787. +
    6788. 62
    6789. »
    6790. diff --git a/doc/pub/week34/html/._week34-bs040.html b/doc/pub/week34/html/._week34-bs040.html index cfbf779dc..ad3664320 100644 --- a/doc/pub/week34/html/._week34-bs040.html +++ b/doc/pub/week34/html/._week34-bs040.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    6791. Installing R, C++, cython or Julia
    6792. Installing R, C++, cython, Numba etc
    6793. Numpy examples and Important Matrix and vector handling packages
    6794. -
    6795. Basic Matrix Features
    6796. -
    6797.    Some famous Matrices
    6798. -
    6799.    More Basic Matrix Features
    6800. -
    6801. Numpy and arrays
    6802. -
    6803. Matrices in Python
    6804. -
    6805. Meet the Pandas
    6806. -
    6807. Friday August 27
    6808. -
    6809.    Simple linear regression model using scikit-learn
    6810. -
    6811.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6812. -
    6813.    Organizing our data
    6814. -
    6815.    Seeing the wood for the trees
    6816. -
    6817.    And what about using neural networks?
    6818. -
    6819. A first summary
    6820. -
    6821. Why Linear Regression (aka Ordinary Least Squares and family)
    6822. -
    6823. Regression analysis, overarching aims
    6824. -
    6825. Regression analysis, overarching aims II
    6826. -
    6827. Examples
    6828. -
    6829. General linear models
    6830. -
    6831. Rewriting the fitting procedure as a linear algebra problem
    6832. -
    6833. Rewriting the fitting procedure as a linear algebra problem, more details
    6834. -
    6835. Generalizing the fitting procedure as a linear algebra problem
    6836. -
    6837. Generalizing the fitting procedure as a linear algebra problem
    6838. -
    6839. Optimizing our parameters
    6840. -
    6841. Our model for the nuclear binding energies
    6842. -
    6843. Optimizing our parameters, more details
    6844. -
    6845. Interpretations and optimizing our parameters
    6846. -
    6847. Interpretations and optimizing our parameters
    6848. -
    6849. Some useful matrix and vector expressions
    6850. -
    6851. Interpretations and optimizing our parameters
    6852. -
    6853. Own code for Ordinary Least Squares
    6854. -
    6855. Adding error analysis and training set up
    6856. -
    6857. The \( \chi^2 \) function
    6858. -
    6859. The \( \chi^2 \) function
    6860. -
    6861. The \( \chi^2 \) function
    6862. -
    6863. The \( \chi^2 \) function
    6864. -
    6865. The \( \chi^2 \) function
    6866. -
    6867. The \( \chi^2 \) function
    6868. -
    6869. Fitting an Equation of State for Dense Nuclear Matter
    6870. -
    6871. The code
    6872. -
    6873. Splitting our Data in Training and Test data
    6874. -
    6875. Exercises
    6876. -
    6877. Exercise 1: Setting up various Python environments
    6878. -
    6879. Exercise 2: making your own data and exploring scikit-learn
    6880. -
    6881. Exercise 3: Normalizing our data
    6882. +
    6883. Numpy and arrays
    6884. +
    6885. Matrices in Python
    6886. +
    6887. Meet the Pandas
    6888. +
    6889.    Simple linear regression model using scikit-learn
    6890. +
    6891.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6892. +
    6893.    Organizing our data
    6894. +
    6895.    And what about using neural networks?
    6896. +
    6897. A first summary
    6898. +
    6899. Why Linear Regression (aka Ordinary Least Squares and family)
    6900. +
    6901. Regression analysis, overarching aims
    6902. +
    6903. Regression analysis, overarching aims II
    6904. +
    6905. Examples
    6906. +
    6907. General linear models
    6908. +
    6909. Rewriting the fitting procedure as a linear algebra problem
    6910. +
    6911. Rewriting the fitting procedure as a linear algebra problem, more details
    6912. +
    6913. Generalizing the fitting procedure as a linear algebra problem
    6914. +
    6915. Generalizing the fitting procedure as a linear algebra problem
    6916. +
    6917. Optimizing our parameters
    6918. +
    6919. Our model for the nuclear binding energies
    6920. +
    6921. Optimizing our parameters, more details
    6922. +
    6923. Interpretations and optimizing our parameters
    6924. +
    6925. Interpretations and optimizing our parameters
    6926. +
    6927. Some useful matrix and vector expressions
    6928. +
    6929. Interpretations and optimizing our parameters
    6930. +
    6931. Own code for Ordinary Least Squares
    6932. +
    6933. Adding error analysis and training set up
    6934. +
    6935. The \( \chi^2 \) function
    6936. +
    6937. The \( \chi^2 \) function
    6938. +
    6939. The \( \chi^2 \) function
    6940. +
    6941. The \( \chi^2 \) function
    6942. +
    6943. The \( \chi^2 \) function
    6944. +
    6945. The \( \chi^2 \) function
    6946. +
    6947. Fitting an Equation of State for Dense Nuclear Matter
    6948. +
    6949. The code
    6950. +
    6951. Splitting our Data in Training and Test data
    6952. +
    6953. Exercises
    6954. +
    6955. Exercise 1: Setting up various Python environments
    6956. +
    6957. Exercise 2: making your own data and exploring scikit-learn
    6958. +
    6959. Exercise 3: Split data in test and training data
    6960. @@ -351,30 +335,43 @@ MathJax.Hub.Config({

       

       

       

      -

      Regression analysis, overarching aims II

      +

      Rewriting the fitting procedure as a linear algebra problem, more details

      +

      Defining the vectors

      +$$ +\boldsymbol{y} = [y_0,y_1, y_2,\dots, y_{n-1}]^T, +$$ -

      Consider an experiment in which \( p \) characteristics of \( n \) samples are -measured. The data from this experiment, for various explanatory variables \( p \) are normally represented by a matrix -\( \mathbf{X} \). -

      +

      and

      +$$ +\boldsymbol{\beta} = [\beta_0,\beta_1, \beta_2,\dots, \beta_{n-1}]^T, +$$ -

      The matrix \( \mathbf{X} \) is called the design -matrix. Additional information of the samples is available in the -form of \( \boldsymbol{y} \) (also as above). The variable \( \boldsymbol{y} \) is -generally referred to as the response variable. The aim of -regression analysis is to explain \( \boldsymbol{y} \) in terms of -\( \boldsymbol{X} \) through a functional relationship like \( y_i = -f(\mathbf{X}_{i,\ast}) \). When no prior knowledge on the form of -\( f(\cdot) \) is available, it is common to assume a linear relationship -between \( \boldsymbol{X} \) and \( \boldsymbol{y} \). This assumption gives rise to -the linear regression model where \( \boldsymbol{\beta} = [\beta_0, \ldots, -\beta_{p-1}]^{T} \) are the regression parameters. -

      +

      and

      +$$ +\boldsymbol{\epsilon} = [\epsilon_0,\epsilon_1, \epsilon_2,\dots, \epsilon_{n-1}]^T, +$$ -

      Linear regression gives us a set of analytical equations for the parameters \( \beta_j \).

      +

      and the design matrix

      +$$ +\boldsymbol{X}= +\begin{bmatrix} +1& x_{0}^1 &x_{0}^2& \dots & \dots &x_{0}^{n-1}\\ +1& x_{1}^1 &x_{1}^2& \dots & \dots &x_{1}^{n-1}\\ +1& x_{2}^1 &x_{2}^2& \dots & \dots &x_{2}^{n-1}\\ +\dots& \dots &\dots& \dots & \dots &\dots\\ +1& x_{n-1}^1 &x_{n-1}^2& \dots & \dots &x_{n-1}^{n-1}\\ +\end{bmatrix} +$$ + +

      we can rewrite our equations as

      +$$ +\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +$$ + +

      The above design matrix is called a Vandermonde matrix.

      @@ -404,7 +401,7 @@ the linear regression model where \( \boldsymbol{\beta} = [\beta_0, \ld
    6961. 49
    6962. 50
    6963. ...
    6964. -
    6965. 66
    6966. +
    6967. 62
    6968. »
    6969. diff --git a/doc/pub/week34/html/._week34-bs041.html b/doc/pub/week34/html/._week34-bs041.html index b729d273c..873636cbc 100644 --- a/doc/pub/week34/html/._week34-bs041.html +++ b/doc/pub/week34/html/._week34-bs041.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    6970. Installing R, C++, cython or Julia
    6971. Installing R, C++, cython, Numba etc
    6972. Numpy examples and Important Matrix and vector handling packages
    6973. -
    6974. Basic Matrix Features
    6975. -
    6976.    Some famous Matrices
    6977. -
    6978.    More Basic Matrix Features
    6979. -
    6980. Numpy and arrays
    6981. -
    6982. Matrices in Python
    6983. -
    6984. Meet the Pandas
    6985. -
    6986. Friday August 27
    6987. -
    6988.    Simple linear regression model using scikit-learn
    6989. -
    6990.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    6991. -
    6992.    Organizing our data
    6993. -
    6994.    Seeing the wood for the trees
    6995. -
    6996.    And what about using neural networks?
    6997. -
    6998. A first summary
    6999. -
    7000. Why Linear Regression (aka Ordinary Least Squares and family)
    7001. -
    7002. Regression analysis, overarching aims
    7003. -
    7004. Regression analysis, overarching aims II
    7005. -
    7006. Examples
    7007. -
    7008. General linear models
    7009. -
    7010. Rewriting the fitting procedure as a linear algebra problem
    7011. -
    7012. Rewriting the fitting procedure as a linear algebra problem, more details
    7013. -
    7014. Generalizing the fitting procedure as a linear algebra problem
    7015. -
    7016. Generalizing the fitting procedure as a linear algebra problem
    7017. -
    7018. Optimizing our parameters
    7019. -
    7020. Our model for the nuclear binding energies
    7021. -
    7022. Optimizing our parameters, more details
    7023. -
    7024. Interpretations and optimizing our parameters
    7025. -
    7026. Interpretations and optimizing our parameters
    7027. -
    7028. Some useful matrix and vector expressions
    7029. -
    7030. Interpretations and optimizing our parameters
    7031. -
    7032. Own code for Ordinary Least Squares
    7033. -
    7034. Adding error analysis and training set up
    7035. -
    7036. The \( \chi^2 \) function
    7037. -
    7038. The \( \chi^2 \) function
    7039. -
    7040. The \( \chi^2 \) function
    7041. -
    7042. The \( \chi^2 \) function
    7043. -
    7044. The \( \chi^2 \) function
    7045. -
    7046. The \( \chi^2 \) function
    7047. -
    7048. Fitting an Equation of State for Dense Nuclear Matter
    7049. -
    7050. The code
    7051. -
    7052. Splitting our Data in Training and Test data
    7053. -
    7054. Exercises
    7055. -
    7056. Exercise 1: Setting up various Python environments
    7057. -
    7058. Exercise 2: making your own data and exploring scikit-learn
    7059. -
    7060. Exercise 3: Normalizing our data
    7061. +
    7062. Numpy and arrays
    7063. +
    7064. Matrices in Python
    7065. +
    7066. Meet the Pandas
    7067. +
    7068.    Simple linear regression model using scikit-learn
    7069. +
    7070.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7071. +
    7072.    Organizing our data
    7073. +
    7074.    And what about using neural networks?
    7075. +
    7076. A first summary
    7077. +
    7078. Why Linear Regression (aka Ordinary Least Squares and family)
    7079. +
    7080. Regression analysis, overarching aims
    7081. +
    7082. Regression analysis, overarching aims II
    7083. +
    7084. Examples
    7085. +
    7086. General linear models
    7087. +
    7088. Rewriting the fitting procedure as a linear algebra problem
    7089. +
    7090. Rewriting the fitting procedure as a linear algebra problem, more details
    7091. +
    7092. Generalizing the fitting procedure as a linear algebra problem
    7093. +
    7094. Generalizing the fitting procedure as a linear algebra problem
    7095. +
    7096. Optimizing our parameters
    7097. +
    7098. Our model for the nuclear binding energies
    7099. +
    7100. Optimizing our parameters, more details
    7101. +
    7102. Interpretations and optimizing our parameters
    7103. +
    7104. Interpretations and optimizing our parameters
    7105. +
    7106. Some useful matrix and vector expressions
    7107. +
    7108. Interpretations and optimizing our parameters
    7109. +
    7110. Own code for Ordinary Least Squares
    7111. +
    7112. Adding error analysis and training set up
    7113. +
    7114. The \( \chi^2 \) function
    7115. +
    7116. The \( \chi^2 \) function
    7117. +
    7118. The \( \chi^2 \) function
    7119. +
    7120. The \( \chi^2 \) function
    7121. +
    7122. The \( \chi^2 \) function
    7123. +
    7124. The \( \chi^2 \) function
    7125. +
    7126. Fitting an Equation of State for Dense Nuclear Matter
    7127. +
    7128. The code
    7129. +
    7130. Splitting our Data in Training and Test data
    7131. +
    7132. Exercises
    7133. +
    7134. Exercise 1: Setting up various Python environments
    7135. +
    7136. Exercise 2: making your own data and exploring scikit-learn
    7137. +
    7138. Exercise 3: Split data in test and training data
    7139. @@ -351,29 +335,32 @@ MathJax.Hub.Config({

       

       

       

      -

      Examples

      +

      Generalizing the fitting procedure as a linear algebra problem

      -

      In order to understand the relation among the predictors \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), -consider the model we discussed for describing nuclear binding energies. + +

      We are obviously not limited to the above polynomial expansions. We +could replace the various powers of \( x \) with elements of Fourier +series or instead of \( x_i^j \) we could have \( \cos{(j x_i)} \) or \( \sin{(j +x_i)} \), or time series or other orthogonal functions. For every set +of values \( y_i,x_i \) we can then generalize the equations to

      -

      There we assumed that we could parametrize the data using a polynomial approximation based on the liquid drop model. -Assuming -

      $$ -BE(A) = a_0+a_1A+a_2A^{2/3}+a_3A^{-1/3}+a_4A^{-1}, +\begin{align*} +y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ +y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ +y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_2\\ +\dots & \dots \\ +y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_i\\ +\dots & \dots \\ +y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ +\end{align*} $$ -

      we have five predictors, that is the intercept, the \( A \) dependent term, the \( A^{2/3} \) term and the \( A^{-1/3} \) and \( A^{-1} \) terms. -This gives \( p=0,1,2,3,4 \). Furthermore we have \( n \) entries for each predictor. It means that our design matrix is a -\( p\times n \) matrix \( \boldsymbol{X} \). -

      -

      Here the predictors are based on a model we have made. A popular data set which is widely encountered in ML applications is the -so-called credit card default data from Taiwan. The data set contains data on \( n=30000 \) credit card holders with predictors like gender, marital status, age, profession, education, etc. In total there are \( 24 \) such predictors or attributes leading to a design matrix of dimensionality \( 24 \times 30000 \). This is however a classification problem and we will come back to it when we discuss Logistic Regression. -

      +Note that we have \( p=n \) here. The matrix is symmetric. This is generally not the case!
      @@ -403,7 +390,7 @@ so-called 50
    7140. 51
    7141. ...
    7142. -
    7143. 66
    7144. +
    7145. 62
    7146. »
    7147. diff --git a/doc/pub/week34/html/._week34-bs042.html b/doc/pub/week34/html/._week34-bs042.html index 20c977b6d..67ddc608d 100644 --- a/doc/pub/week34/html/._week34-bs042.html +++ b/doc/pub/week34/html/._week34-bs042.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    7148. Installing R, C++, cython or Julia
    7149. Installing R, C++, cython, Numba etc
    7150. Numpy examples and Important Matrix and vector handling packages
    7151. -
    7152. Basic Matrix Features
    7153. -
    7154.    Some famous Matrices
    7155. -
    7156.    More Basic Matrix Features
    7157. -
    7158. Numpy and arrays
    7159. -
    7160. Matrices in Python
    7161. -
    7162. Meet the Pandas
    7163. -
    7164. Friday August 27
    7165. -
    7166.    Simple linear regression model using scikit-learn
    7167. -
    7168.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7169. -
    7170.    Organizing our data
    7171. -
    7172.    Seeing the wood for the trees
    7173. -
    7174.    And what about using neural networks?
    7175. -
    7176. A first summary
    7177. -
    7178. Why Linear Regression (aka Ordinary Least Squares and family)
    7179. -
    7180. Regression analysis, overarching aims
    7181. -
    7182. Regression analysis, overarching aims II
    7183. -
    7184. Examples
    7185. -
    7186. General linear models
    7187. -
    7188. Rewriting the fitting procedure as a linear algebra problem
    7189. -
    7190. Rewriting the fitting procedure as a linear algebra problem, more details
    7191. -
    7192. Generalizing the fitting procedure as a linear algebra problem
    7193. -
    7194. Generalizing the fitting procedure as a linear algebra problem
    7195. -
    7196. Optimizing our parameters
    7197. -
    7198. Our model for the nuclear binding energies
    7199. -
    7200. Optimizing our parameters, more details
    7201. -
    7202. Interpretations and optimizing our parameters
    7203. -
    7204. Interpretations and optimizing our parameters
    7205. -
    7206. Some useful matrix and vector expressions
    7207. -
    7208. Interpretations and optimizing our parameters
    7209. -
    7210. Own code for Ordinary Least Squares
    7211. -
    7212. Adding error analysis and training set up
    7213. -
    7214. The \( \chi^2 \) function
    7215. -
    7216. The \( \chi^2 \) function
    7217. -
    7218. The \( \chi^2 \) function
    7219. -
    7220. The \( \chi^2 \) function
    7221. -
    7222. The \( \chi^2 \) function
    7223. -
    7224. The \( \chi^2 \) function
    7225. -
    7226. Fitting an Equation of State for Dense Nuclear Matter
    7227. -
    7228. The code
    7229. -
    7230. Splitting our Data in Training and Test data
    7231. -
    7232. Exercises
    7233. -
    7234. Exercise 1: Setting up various Python environments
    7235. -
    7236. Exercise 2: making your own data and exploring scikit-learn
    7237. -
    7238. Exercise 3: Normalizing our data
    7239. +
    7240. Numpy and arrays
    7241. +
    7242. Matrices in Python
    7243. +
    7244. Meet the Pandas
    7245. +
    7246.    Simple linear regression model using scikit-learn
    7247. +
    7248.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7249. +
    7250.    Organizing our data
    7251. +
    7252.    And what about using neural networks?
    7253. +
    7254. A first summary
    7255. +
    7256. Why Linear Regression (aka Ordinary Least Squares and family)
    7257. +
    7258. Regression analysis, overarching aims
    7259. +
    7260. Regression analysis, overarching aims II
    7261. +
    7262. Examples
    7263. +
    7264. General linear models
    7265. +
    7266. Rewriting the fitting procedure as a linear algebra problem
    7267. +
    7268. Rewriting the fitting procedure as a linear algebra problem, more details
    7269. +
    7270. Generalizing the fitting procedure as a linear algebra problem
    7271. +
    7272. Generalizing the fitting procedure as a linear algebra problem
    7273. +
    7274. Optimizing our parameters
    7275. +
    7276. Our model for the nuclear binding energies
    7277. +
    7278. Optimizing our parameters, more details
    7279. +
    7280. Interpretations and optimizing our parameters
    7281. +
    7282. Interpretations and optimizing our parameters
    7283. +
    7284. Some useful matrix and vector expressions
    7285. +
    7286. Interpretations and optimizing our parameters
    7287. +
    7288. Own code for Ordinary Least Squares
    7289. +
    7290. Adding error analysis and training set up
    7291. +
    7292. The \( \chi^2 \) function
    7293. +
    7294. The \( \chi^2 \) function
    7295. +
    7296. The \( \chi^2 \) function
    7297. +
    7298. The \( \chi^2 \) function
    7299. +
    7300. The \( \chi^2 \) function
    7301. +
    7302. The \( \chi^2 \) function
    7303. +
    7304. Fitting an Equation of State for Dense Nuclear Matter
    7305. +
    7306. The code
    7307. +
    7308. Splitting our Data in Training and Test data
    7309. +
    7310. Exercises
    7311. +
    7312. Exercise 1: Setting up various Python environments
    7313. +
    7314. Exercise 2: making your own data and exploring scikit-learn
    7315. +
    7316. Exercise 3: Split data in test and training data
    7317. @@ -351,18 +335,28 @@ MathJax.Hub.Config({

       

       

       

      -

      General linear models

      +

      Generalizing the fitting procedure as a linear algebra problem

      -

      Before we proceed let us study a case from linear algebra where we aim at fitting a set of data \( \boldsymbol{y}=[y_0,y_1,\dots,y_{n-1}] \). We could think of these data as a result of an experiment or a complicated numerical experiment. These data are functions of a series of variables \( \boldsymbol{x}=[x_0,x_1,\dots,x_{n-1}] \), that is \( y_i = y(x_i) \) with \( i=0,1,2,\dots,n-1 \). The variables \( x_i \) could represent physical quantities like time, temperature, position etc. We assume that \( y(x) \) is a smooth function.

      - -

      Since obtaining these data points may not be trivial, we want to use these data to fit a function which can allow us to make predictions for values of \( y \) which are not in the present set. The perhaps simplest approach is to assume we can parametrize our function in terms of a polynomial of degree \( n-1 \) with \( n \) points, that is

      +

      We redefine in turn the matrix \( \boldsymbol{X} \) as

      $$ -y=y(x) \rightarrow y(x_i)=\tilde{y}_i+\epsilon_i=\sum_{j=0}^{n-1} \beta_j x_i^j+\epsilon_i, +\boldsymbol{X}= +\begin{bmatrix} +x_{00}& x_{01} &x_{02}& \dots & \dots &x_{0,n-1}\\ +x_{10}& x_{11} &x_{12}& \dots & \dots &x_{1,n-1}\\ +x_{20}& x_{21} &x_{22}& \dots & \dots &x_{2,n-1}\\ +\dots& \dots &\dots& \dots & \dots &\dots\\ +x_{n-1,0}& x_{n-1,1} &x_{n-1,2}& \dots & \dots &x_{n-1,n-1}\\ +\end{bmatrix} $$ -

      where \( \epsilon_i \) is the error in our approximation.

      +

      and without loss of generality we rewrite again our equations as

      +$$ +\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +$$ + +

      The left-hand side of this equation is kwown. Our error vector \( \boldsymbol{\epsilon} \) and the parameter vector \( \boldsymbol{\beta} \) are our unknow quantities. How can we obtain the optimal set of \( \beta_i \) values?

      @@ -392,7 +386,7 @@ $$
    7318. 51
    7319. 52
    7320. ...
    7321. -
    7322. 66
    7323. +
    7324. 62
    7325. »
    7326. diff --git a/doc/pub/week34/html/._week34-bs043.html b/doc/pub/week34/html/._week34-bs043.html index 927066a19..f49f21f94 100644 --- a/doc/pub/week34/html/._week34-bs043.html +++ b/doc/pub/week34/html/._week34-bs043.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    7327. Installing R, C++, cython or Julia
    7328. Installing R, C++, cython, Numba etc
    7329. Numpy examples and Important Matrix and vector handling packages
    7330. -
    7331. Basic Matrix Features
    7332. -
    7333.    Some famous Matrices
    7334. -
    7335.    More Basic Matrix Features
    7336. -
    7337. Numpy and arrays
    7338. -
    7339. Matrices in Python
    7340. -
    7341. Meet the Pandas
    7342. -
    7343. Friday August 27
    7344. -
    7345.    Simple linear regression model using scikit-learn
    7346. -
    7347.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7348. -
    7349.    Organizing our data
    7350. -
    7351.    Seeing the wood for the trees
    7352. -
    7353.    And what about using neural networks?
    7354. -
    7355. A first summary
    7356. -
    7357. Why Linear Regression (aka Ordinary Least Squares and family)
    7358. -
    7359. Regression analysis, overarching aims
    7360. -
    7361. Regression analysis, overarching aims II
    7362. -
    7363. Examples
    7364. -
    7365. General linear models
    7366. -
    7367. Rewriting the fitting procedure as a linear algebra problem
    7368. -
    7369. Rewriting the fitting procedure as a linear algebra problem, more details
    7370. -
    7371. Generalizing the fitting procedure as a linear algebra problem
    7372. -
    7373. Generalizing the fitting procedure as a linear algebra problem
    7374. -
    7375. Optimizing our parameters
    7376. -
    7377. Our model for the nuclear binding energies
    7378. -
    7379. Optimizing our parameters, more details
    7380. -
    7381. Interpretations and optimizing our parameters
    7382. -
    7383. Interpretations and optimizing our parameters
    7384. -
    7385. Some useful matrix and vector expressions
    7386. -
    7387. Interpretations and optimizing our parameters
    7388. -
    7389. Own code for Ordinary Least Squares
    7390. -
    7391. Adding error analysis and training set up
    7392. -
    7393. The \( \chi^2 \) function
    7394. -
    7395. The \( \chi^2 \) function
    7396. -
    7397. The \( \chi^2 \) function
    7398. -
    7399. The \( \chi^2 \) function
    7400. -
    7401. The \( \chi^2 \) function
    7402. -
    7403. The \( \chi^2 \) function
    7404. -
    7405. Fitting an Equation of State for Dense Nuclear Matter
    7406. -
    7407. The code
    7408. -
    7409. Splitting our Data in Training and Test data
    7410. -
    7411. Exercises
    7412. -
    7413. Exercise 1: Setting up various Python environments
    7414. -
    7415. Exercise 2: making your own data and exploring scikit-learn
    7416. -
    7417. Exercise 3: Normalizing our data
    7418. +
    7419. Numpy and arrays
    7420. +
    7421. Matrices in Python
    7422. +
    7423. Meet the Pandas
    7424. +
    7425.    Simple linear regression model using scikit-learn
    7426. +
    7427.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7428. +
    7429.    Organizing our data
    7430. +
    7431.    And what about using neural networks?
    7432. +
    7433. A first summary
    7434. +
    7435. Why Linear Regression (aka Ordinary Least Squares and family)
    7436. +
    7437. Regression analysis, overarching aims
    7438. +
    7439. Regression analysis, overarching aims II
    7440. +
    7441. Examples
    7442. +
    7443. General linear models
    7444. +
    7445. Rewriting the fitting procedure as a linear algebra problem
    7446. +
    7447. Rewriting the fitting procedure as a linear algebra problem, more details
    7448. +
    7449. Generalizing the fitting procedure as a linear algebra problem
    7450. +
    7451. Generalizing the fitting procedure as a linear algebra problem
    7452. +
    7453. Optimizing our parameters
    7454. +
    7455. Our model for the nuclear binding energies
    7456. +
    7457. Optimizing our parameters, more details
    7458. +
    7459. Interpretations and optimizing our parameters
    7460. +
    7461. Interpretations and optimizing our parameters
    7462. +
    7463. Some useful matrix and vector expressions
    7464. +
    7465. Interpretations and optimizing our parameters
    7466. +
    7467. Own code for Ordinary Least Squares
    7468. +
    7469. Adding error analysis and training set up
    7470. +
    7471. The \( \chi^2 \) function
    7472. +
    7473. The \( \chi^2 \) function
    7474. +
    7475. The \( \chi^2 \) function
    7476. +
    7477. The \( \chi^2 \) function
    7478. +
    7479. The \( \chi^2 \) function
    7480. +
    7481. The \( \chi^2 \) function
    7482. +
    7483. Fitting an Equation of State for Dense Nuclear Matter
    7484. +
    7485. The code
    7486. +
    7487. Splitting our Data in Training and Test data
    7488. +
    7489. Exercises
    7490. +
    7491. Exercise 1: Setting up various Python environments
    7492. +
    7493. Exercise 2: making your own data and exploring scikit-learn
    7494. +
    7495. Exercise 3: Split data in test and training data
    7496. @@ -351,20 +335,27 @@ MathJax.Hub.Config({

       

       

       

      -

      Rewriting the fitting procedure as a linear algebra problem

      +

      Optimizing our parameters

      -

      For every set of values \( y_i,x_i \) we have thus the corresponding set of equations

      +

      We have defined the matrix \( \boldsymbol{X} \) via the equations

      $$ \begin{align*} -y_0&=\beta_0+\beta_1x_0^1+\beta_2x_0^2+\dots+\beta_{n-1}x_0^{n-1}+\epsilon_0\\ -y_1&=\beta_0+\beta_1x_1^1+\beta_2x_1^2+\dots+\beta_{n-1}x_1^{n-1}+\epsilon_1\\ -y_2&=\beta_0+\beta_1x_2^1+\beta_2x_2^2+\dots+\beta_{n-1}x_2^{n-1}+\epsilon_2\\ +y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ +y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ +y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_1\\ \dots & \dots \\ -y_{n-1}&=\beta_0+\beta_1x_{n-1}^1+\beta_2x_{n-1}^2+\dots+\beta_{n-1}x_{n-1}^{n-1}+\epsilon_{n-1}.\\ +y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_1\\ +\dots & \dots \\ +y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ \end{align*} $$ + +

      As we noted above, we stayed with a system with the design matrix + \( \boldsymbol{X}\in {\mathbb{R}}^{n\times n} \), that is we have \( p=n \). For reasons to come later (algorithmic arguments) we will hereafter define +our matrix as \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \), with the predictors refering to the column numbers and the entries \( n \) being the row elements. +

      @@ -394,7 +385,7 @@ $$
    7497. 52
    7498. 53
    7499. ...
    7500. -
    7501. 66
    7502. +
    7503. 62
    7504. »
    7505. diff --git a/doc/pub/week34/html/._week34-bs044.html b/doc/pub/week34/html/._week34-bs044.html index a687ae449..72fcd2b11 100644 --- a/doc/pub/week34/html/._week34-bs044.html +++ b/doc/pub/week34/html/._week34-bs044.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    7506. Installing R, C++, cython or Julia
    7507. Installing R, C++, cython, Numba etc
    7508. Numpy examples and Important Matrix and vector handling packages
    7509. -
    7510. Basic Matrix Features
    7511. -
    7512.    Some famous Matrices
    7513. -
    7514.    More Basic Matrix Features
    7515. -
    7516. Numpy and arrays
    7517. -
    7518. Matrices in Python
    7519. -
    7520. Meet the Pandas
    7521. -
    7522. Friday August 27
    7523. -
    7524.    Simple linear regression model using scikit-learn
    7525. -
    7526.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7527. -
    7528.    Organizing our data
    7529. -
    7530.    Seeing the wood for the trees
    7531. -
    7532.    And what about using neural networks?
    7533. -
    7534. A first summary
    7535. -
    7536. Why Linear Regression (aka Ordinary Least Squares and family)
    7537. -
    7538. Regression analysis, overarching aims
    7539. -
    7540. Regression analysis, overarching aims II
    7541. -
    7542. Examples
    7543. -
    7544. General linear models
    7545. -
    7546. Rewriting the fitting procedure as a linear algebra problem
    7547. -
    7548. Rewriting the fitting procedure as a linear algebra problem, more details
    7549. -
    7550. Generalizing the fitting procedure as a linear algebra problem
    7551. -
    7552. Generalizing the fitting procedure as a linear algebra problem
    7553. -
    7554. Optimizing our parameters
    7555. -
    7556. Our model for the nuclear binding energies
    7557. -
    7558. Optimizing our parameters, more details
    7559. -
    7560. Interpretations and optimizing our parameters
    7561. -
    7562. Interpretations and optimizing our parameters
    7563. -
    7564. Some useful matrix and vector expressions
    7565. -
    7566. Interpretations and optimizing our parameters
    7567. -
    7568. Own code for Ordinary Least Squares
    7569. -
    7570. Adding error analysis and training set up
    7571. -
    7572. The \( \chi^2 \) function
    7573. -
    7574. The \( \chi^2 \) function
    7575. -
    7576. The \( \chi^2 \) function
    7577. -
    7578. The \( \chi^2 \) function
    7579. -
    7580. The \( \chi^2 \) function
    7581. -
    7582. The \( \chi^2 \) function
    7583. -
    7584. Fitting an Equation of State for Dense Nuclear Matter
    7585. -
    7586. The code
    7587. -
    7588. Splitting our Data in Training and Test data
    7589. -
    7590. Exercises
    7591. -
    7592. Exercise 1: Setting up various Python environments
    7593. -
    7594. Exercise 2: making your own data and exploring scikit-learn
    7595. -
    7596. Exercise 3: Normalizing our data
    7597. +
    7598. Numpy and arrays
    7599. +
    7600. Matrices in Python
    7601. +
    7602. Meet the Pandas
    7603. +
    7604.    Simple linear regression model using scikit-learn
    7605. +
    7606.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7607. +
    7608.    Organizing our data
    7609. +
    7610.    And what about using neural networks?
    7611. +
    7612. A first summary
    7613. +
    7614. Why Linear Regression (aka Ordinary Least Squares and family)
    7615. +
    7616. Regression analysis, overarching aims
    7617. +
    7618. Regression analysis, overarching aims II
    7619. +
    7620. Examples
    7621. +
    7622. General linear models
    7623. +
    7624. Rewriting the fitting procedure as a linear algebra problem
    7625. +
    7626. Rewriting the fitting procedure as a linear algebra problem, more details
    7627. +
    7628. Generalizing the fitting procedure as a linear algebra problem
    7629. +
    7630. Generalizing the fitting procedure as a linear algebra problem
    7631. +
    7632. Optimizing our parameters
    7633. +
    7634. Our model for the nuclear binding energies
    7635. +
    7636. Optimizing our parameters, more details
    7637. +
    7638. Interpretations and optimizing our parameters
    7639. +
    7640. Interpretations and optimizing our parameters
    7641. +
    7642. Some useful matrix and vector expressions
    7643. +
    7644. Interpretations and optimizing our parameters
    7645. +
    7646. Own code for Ordinary Least Squares
    7647. +
    7648. Adding error analysis and training set up
    7649. +
    7650. The \( \chi^2 \) function
    7651. +
    7652. The \( \chi^2 \) function
    7653. +
    7654. The \( \chi^2 \) function
    7655. +
    7656. The \( \chi^2 \) function
    7657. +
    7658. The \( \chi^2 \) function
    7659. +
    7660. The \( \chi^2 \) function
    7661. +
    7662. Fitting an Equation of State for Dense Nuclear Matter
    7663. +
    7664. The code
    7665. +
    7666. Splitting our Data in Training and Test data
    7667. +
    7668. Exercises
    7669. +
    7670. Exercise 1: Setting up various Python environments
    7671. +
    7672. Exercise 2: making your own data and exploring scikit-learn
    7673. +
    7674. Exercise 3: Split data in test and training data
    7675. @@ -351,46 +335,108 @@ MathJax.Hub.Config({

       

       

       

      -

      Rewriting the fitting procedure as a linear algebra problem, more details

      -
      -
      - -

      Defining the vectors

      -$$ -\boldsymbol{y} = [y_0,y_1, y_2,\dots, y_{n-1}]^T, -$$ +

      Our model for the nuclear binding energies

      -

      and

      -$$ -\boldsymbol{\beta} = [\beta_0,\beta_1, \beta_2,\dots, \beta_{n-1}]^T, -$$ +

      In our introductory notes we looked at the so-called liquid drop model. Let us remind ourselves about what we did by looking at the code.

      -

      and

      -$$ -\boldsymbol{\epsilon} = [\epsilon_0,\epsilon_1, \epsilon_2,\dots, \epsilon_{n-1}]^T, -$$ +

      We restate the parts of the code we are most interested in.

      -

      and the design matrix

      -$$ -\boldsymbol{X}= -\begin{bmatrix} -1& x_{0}^1 &x_{0}^2& \dots & \dots &x_{0}^{n-1}\\ -1& x_{1}^1 &x_{1}^2& \dots & \dots &x_{1}^{n-1}\\ -1& x_{2}^1 &x_{2}^2& \dots & \dots &x_{2}^{n-1}\\ -\dots& \dots &\dots& \dots & \dots &\dots\\ -1& x_{n-1}^1 &x_{n-1}^2& \dots & \dots &x_{n-1}^{n-1}\\ -\end{bmatrix} -$$ + +
      +
      +
      +
      +
      +
      # Common imports
      +import numpy as np
      +import pandas as pd
      +import matplotlib.pyplot as plt
      +from IPython.display import display
      +import os
       
      -

      we can rewrite our equations as

      -$$ -\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. -$$ +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" -

      The above design matrix is called a Vandermonde matrix.

      +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("MassEval2016.dat"),'r') + + +# Read the experimental data with Pandas +Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11), + names=('N', 'Z', 'A', 'Element', 'Ebinding'), + widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1), + header=39, + index_col=False) + +# Extrapolated values are indicated by '#' in place of the decimal place, so +# the Ebinding column won't be numeric. Coerce to float and drop these entries. +Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce') +Masses = Masses.dropna() +# Convert from keV to MeV. +Masses['Ebinding'] /= 1000 + +# Group the DataFrame by nucleon number, A. +Masses = Masses.groupby('A') +# Find the rows of the grouped DataFrame with the maximum binding energy. +Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()]) +A = Masses['A'] +Z = Masses['Z'] +N = Masses['N'] +Element = Masses['Element'] +Energies = Masses['Ebinding'] + +# Now we set up the design matrix X +X = np.zeros((len(A),5)) +X[:,0] = 1 +X[:,1] = A +X[:,2] = A**(2.0/3.0) +X[:,3] = A**(-1.0/3.0) +X[:,4] = A**(-1.0) +# Then nice printout using pandas +DesignMatrix = pd.DataFrame(X) +DesignMatrix.index = A +DesignMatrix.columns = ['1', 'A', 'A^(2/3)', 'A^(-1/3)', '1/A'] +display(DesignMatrix) +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +

      With \( \boldsymbol{\beta}\in {\mathbb{R}}^{p\times 1} \), it means that we will hereafter write our equations for the approximation as

      +$$ +\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +$$ + +

      throughout these lectures.

      @@ -417,7 +463,7 @@ $$

    7676. 53
    7677. 54
    7678. ...
    7679. -
    7680. 66
    7681. +
    7682. 62
    7683. »
    7684. diff --git a/doc/pub/week34/html/._week34-bs045.html b/doc/pub/week34/html/._week34-bs045.html index 89bddfcc3..299eee68f 100644 --- a/doc/pub/week34/html/._week34-bs045.html +++ b/doc/pub/week34/html/._week34-bs045.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    7685. Installing R, C++, cython or Julia
    7686. Installing R, C++, cython, Numba etc
    7687. Numpy examples and Important Matrix and vector handling packages
    7688. -
    7689. Basic Matrix Features
    7690. -
    7691.    Some famous Matrices
    7692. -
    7693.    More Basic Matrix Features
    7694. -
    7695. Numpy and arrays
    7696. -
    7697. Matrices in Python
    7698. -
    7699. Meet the Pandas
    7700. -
    7701. Friday August 27
    7702. -
    7703.    Simple linear regression model using scikit-learn
    7704. -
    7705.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7706. -
    7707.    Organizing our data
    7708. -
    7709.    Seeing the wood for the trees
    7710. -
    7711.    And what about using neural networks?
    7712. -
    7713. A first summary
    7714. -
    7715. Why Linear Regression (aka Ordinary Least Squares and family)
    7716. -
    7717. Regression analysis, overarching aims
    7718. -
    7719. Regression analysis, overarching aims II
    7720. -
    7721. Examples
    7722. -
    7723. General linear models
    7724. -
    7725. Rewriting the fitting procedure as a linear algebra problem
    7726. -
    7727. Rewriting the fitting procedure as a linear algebra problem, more details
    7728. -
    7729. Generalizing the fitting procedure as a linear algebra problem
    7730. -
    7731. Generalizing the fitting procedure as a linear algebra problem
    7732. -
    7733. Optimizing our parameters
    7734. -
    7735. Our model for the nuclear binding energies
    7736. -
    7737. Optimizing our parameters, more details
    7738. -
    7739. Interpretations and optimizing our parameters
    7740. -
    7741. Interpretations and optimizing our parameters
    7742. -
    7743. Some useful matrix and vector expressions
    7744. -
    7745. Interpretations and optimizing our parameters
    7746. -
    7747. Own code for Ordinary Least Squares
    7748. -
    7749. Adding error analysis and training set up
    7750. -
    7751. The \( \chi^2 \) function
    7752. -
    7753. The \( \chi^2 \) function
    7754. -
    7755. The \( \chi^2 \) function
    7756. -
    7757. The \( \chi^2 \) function
    7758. -
    7759. The \( \chi^2 \) function
    7760. -
    7761. The \( \chi^2 \) function
    7762. -
    7763. Fitting an Equation of State for Dense Nuclear Matter
    7764. -
    7765. The code
    7766. -
    7767. Splitting our Data in Training and Test data
    7768. -
    7769. Exercises
    7770. -
    7771. Exercise 1: Setting up various Python environments
    7772. -
    7773. Exercise 2: making your own data and exploring scikit-learn
    7774. -
    7775. Exercise 3: Normalizing our data
    7776. +
    7777. Numpy and arrays
    7778. +
    7779. Matrices in Python
    7780. +
    7781. Meet the Pandas
    7782. +
    7783.    Simple linear regression model using scikit-learn
    7784. +
    7785.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7786. +
    7787.    Organizing our data
    7788. +
    7789.    And what about using neural networks?
    7790. +
    7791. A first summary
    7792. +
    7793. Why Linear Regression (aka Ordinary Least Squares and family)
    7794. +
    7795. Regression analysis, overarching aims
    7796. +
    7797. Regression analysis, overarching aims II
    7798. +
    7799. Examples
    7800. +
    7801. General linear models
    7802. +
    7803. Rewriting the fitting procedure as a linear algebra problem
    7804. +
    7805. Rewriting the fitting procedure as a linear algebra problem, more details
    7806. +
    7807. Generalizing the fitting procedure as a linear algebra problem
    7808. +
    7809. Generalizing the fitting procedure as a linear algebra problem
    7810. +
    7811. Optimizing our parameters
    7812. +
    7813. Our model for the nuclear binding energies
    7814. +
    7815. Optimizing our parameters, more details
    7816. +
    7817. Interpretations and optimizing our parameters
    7818. +
    7819. Interpretations and optimizing our parameters
    7820. +
    7821. Some useful matrix and vector expressions
    7822. +
    7823. Interpretations and optimizing our parameters
    7824. +
    7825. Own code for Ordinary Least Squares
    7826. +
    7827. Adding error analysis and training set up
    7828. +
    7829. The \( \chi^2 \) function
    7830. +
    7831. The \( \chi^2 \) function
    7832. +
    7833. The \( \chi^2 \) function
    7834. +
    7835. The \( \chi^2 \) function
    7836. +
    7837. The \( \chi^2 \) function
    7838. +
    7839. The \( \chi^2 \) function
    7840. +
    7841. Fitting an Equation of State for Dense Nuclear Matter
    7842. +
    7843. The code
    7844. +
    7845. Splitting our Data in Training and Test data
    7846. +
    7847. Exercises
    7848. +
    7849. Exercise 1: Setting up various Python environments
    7850. +
    7851. Exercise 2: making your own data and exploring scikit-learn
    7852. +
    7853. Exercise 3: Split data in test and training data
    7854. @@ -351,32 +335,36 @@ MathJax.Hub.Config({

       

       

       

      -

      Generalizing the fitting procedure as a linear algebra problem

      +

      Optimizing our parameters, more details

      +

      With the above we use the design matrix to define the approximation \( \boldsymbol{\tilde{y}} \) via the unknown quantity \( \boldsymbol{\beta} \) as

      +$$ +\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +$$ -

      We are obviously not limited to the above polynomial expansions. We -could replace the various powers of \( x \) with elements of Fourier -series or instead of \( x_i^j \) we could have \( \cos{(j x_i)} \) or \( \sin{(j -x_i)} \), or time series or other orthogonal functions. For every set -of values \( y_i,x_i \) we can then generalize the equations to +

      and in order to find the optimal parameters \( \beta_i \) instead of solving the above linear algebra problem, we define a function which gives a measure of the spread between the values \( y_i \) (which represent hopefully the exact values) and the parameterized values \( \tilde{y}_i \), namely

      +$$ +C(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +$$ + +

      or using the matrix \( \boldsymbol{X} \) and in a more compact matrix-vector notation as

      +$$ +C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +$$ + +

      This function is one possible way to define the so-called cost function.

      + +

      It is also common to define +the function \( C \) as

      $$ -\begin{align*} -y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ -y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ -y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_2\\ -\dots & \dots \\ -y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_i\\ -\dots & \dots \\ -y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ -\end{align*} +C(\boldsymbol{\beta})=\frac{1}{2n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2, $$ - -Note that we have \( p=n \) here. The matrix is symmetric. This is generally not the case! +

      since when taking the first derivative with respect to the unknown parameters \( \beta \), the factor of \( 2 \) cancels out.

      @@ -406,7 +394,7 @@ $$
    7855. 54
    7856. 55
    7857. ...
    7858. -
    7859. 66
    7860. +
    7861. 62
    7862. »
    7863. diff --git a/doc/pub/week34/html/._week34-bs046.html b/doc/pub/week34/html/._week34-bs046.html index e7ea2dc15..dc96babec 100644 --- a/doc/pub/week34/html/._week34-bs046.html +++ b/doc/pub/week34/html/._week34-bs046.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    7864. Installing R, C++, cython or Julia
    7865. Installing R, C++, cython, Numba etc
    7866. Numpy examples and Important Matrix and vector handling packages
    7867. -
    7868. Basic Matrix Features
    7869. -
    7870.    Some famous Matrices
    7871. -
    7872.    More Basic Matrix Features
    7873. -
    7874. Numpy and arrays
    7875. -
    7876. Matrices in Python
    7877. -
    7878. Meet the Pandas
    7879. -
    7880. Friday August 27
    7881. -
    7882.    Simple linear regression model using scikit-learn
    7883. -
    7884.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7885. -
    7886.    Organizing our data
    7887. -
    7888.    Seeing the wood for the trees
    7889. -
    7890.    And what about using neural networks?
    7891. -
    7892. A first summary
    7893. -
    7894. Why Linear Regression (aka Ordinary Least Squares and family)
    7895. -
    7896. Regression analysis, overarching aims
    7897. -
    7898. Regression analysis, overarching aims II
    7899. -
    7900. Examples
    7901. -
    7902. General linear models
    7903. -
    7904. Rewriting the fitting procedure as a linear algebra problem
    7905. -
    7906. Rewriting the fitting procedure as a linear algebra problem, more details
    7907. -
    7908. Generalizing the fitting procedure as a linear algebra problem
    7909. -
    7910. Generalizing the fitting procedure as a linear algebra problem
    7911. -
    7912. Optimizing our parameters
    7913. -
    7914. Our model for the nuclear binding energies
    7915. -
    7916. Optimizing our parameters, more details
    7917. -
    7918. Interpretations and optimizing our parameters
    7919. -
    7920. Interpretations and optimizing our parameters
    7921. -
    7922. Some useful matrix and vector expressions
    7923. -
    7924. Interpretations and optimizing our parameters
    7925. -
    7926. Own code for Ordinary Least Squares
    7927. -
    7928. Adding error analysis and training set up
    7929. -
    7930. The \( \chi^2 \) function
    7931. -
    7932. The \( \chi^2 \) function
    7933. -
    7934. The \( \chi^2 \) function
    7935. -
    7936. The \( \chi^2 \) function
    7937. -
    7938. The \( \chi^2 \) function
    7939. -
    7940. The \( \chi^2 \) function
    7941. -
    7942. Fitting an Equation of State for Dense Nuclear Matter
    7943. -
    7944. The code
    7945. -
    7946. Splitting our Data in Training and Test data
    7947. -
    7948. Exercises
    7949. -
    7950. Exercise 1: Setting up various Python environments
    7951. -
    7952. Exercise 2: making your own data and exploring scikit-learn
    7953. -
    7954. Exercise 3: Normalizing our data
    7955. +
    7956. Numpy and arrays
    7957. +
    7958. Matrices in Python
    7959. +
    7960. Meet the Pandas
    7961. +
    7962.    Simple linear regression model using scikit-learn
    7963. +
    7964.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    7965. +
    7966.    Organizing our data
    7967. +
    7968.    And what about using neural networks?
    7969. +
    7970. A first summary
    7971. +
    7972. Why Linear Regression (aka Ordinary Least Squares and family)
    7973. +
    7974. Regression analysis, overarching aims
    7975. +
    7976. Regression analysis, overarching aims II
    7977. +
    7978. Examples
    7979. +
    7980. General linear models
    7981. +
    7982. Rewriting the fitting procedure as a linear algebra problem
    7983. +
    7984. Rewriting the fitting procedure as a linear algebra problem, more details
    7985. +
    7986. Generalizing the fitting procedure as a linear algebra problem
    7987. +
    7988. Generalizing the fitting procedure as a linear algebra problem
    7989. +
    7990. Optimizing our parameters
    7991. +
    7992. Our model for the nuclear binding energies
    7993. +
    7994. Optimizing our parameters, more details
    7995. +
    7996. Interpretations and optimizing our parameters
    7997. +
    7998. Interpretations and optimizing our parameters
    7999. +
    8000. Some useful matrix and vector expressions
    8001. +
    8002. Interpretations and optimizing our parameters
    8003. +
    8004. Own code for Ordinary Least Squares
    8005. +
    8006. Adding error analysis and training set up
    8007. +
    8008. The \( \chi^2 \) function
    8009. +
    8010. The \( \chi^2 \) function
    8011. +
    8012. The \( \chi^2 \) function
    8013. +
    8014. The \( \chi^2 \) function
    8015. +
    8016. The \( \chi^2 \) function
    8017. +
    8018. The \( \chi^2 \) function
    8019. +
    8020. Fitting an Equation of State for Dense Nuclear Matter
    8021. +
    8022. The code
    8023. +
    8024. Splitting our Data in Training and Test data
    8025. +
    8026. Exercises
    8027. +
    8028. Exercise 1: Setting up various Python environments
    8029. +
    8030. Exercise 2: making your own data and exploring scikit-learn
    8031. +
    8032. Exercise 3: Split data in test and training data
    8033. @@ -351,28 +335,53 @@ MathJax.Hub.Config({

       

       

       

      -

      Generalizing the fitting procedure as a linear algebra problem

      +

      Interpretations and optimizing our parameters

      -

      We redefine in turn the matrix \( \boldsymbol{X} \) as

      + +

      The function

      $$ -\boldsymbol{X}= -\begin{bmatrix} -x_{00}& x_{01} &x_{02}& \dots & \dots &x_{0,n-1}\\ -x_{10}& x_{11} &x_{12}& \dots & \dots &x_{1,n-1}\\ -x_{20}& x_{21} &x_{22}& \dots & \dots &x_{2,n-1}\\ -\dots& \dots &\dots& \dots & \dots &\dots\\ -x_{n-1,0}& x_{n-1,1} &x_{n-1,2}& \dots & \dots &x_{n-1,n-1}\\ -\end{bmatrix} +C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}, $$ -

      and without loss of generality we rewrite again our equations as

      +

      can be linked to the variance of the quantity \( y_i \) if we interpret the latter as the mean value. +When linking (see the discussion below) with the maximum likelihood approach below, we will indeed interpret \( y_i \) as a mean value +

      $$ -\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +y_{i}=\langle y_i \rangle = \beta_0x_{i,0}+\beta_1x_{i,1}+\beta_2x_{i,2}+\dots+\beta_{n-1}x_{i,n-1}+\epsilon_i, $$ -

      The left-hand side of this equation is kwown. Our error vector \( \boldsymbol{\epsilon} \) and the parameter vector \( \boldsymbol{\beta} \) are our unknow quantities. How can we obtain the optimal set of \( \beta_i \) values?

      +

      where \( \langle y_i \rangle \) is the mean value. Keep in mind also that +till now we have treated \( y_i \) as the exact value. Normally, the +response (dependent or outcome) variable \( y_i \) the outcome of a +numerical experiment or another type of experiment and is thus only an +approximation to the true value. It is then always accompanied by an +error estimate, often limited to a statistical error estimate given by +the standard deviation discussed earlier. In the discussion here we +will treat \( y_i \) as our exact value for the response variable. +

      + +

      In order to find the parameters \( \beta_i \) we will then minimize the spread of \( C(\boldsymbol{\beta}) \), that is we are going to solve the problem

      +$$ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +$$ + +

      In practical terms it means we will require

      +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)^2\right]=0, +$$ + +

      which results in

      +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_{ij}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)\right]=0, +$$ + +

      or in a matrix-vector form as

      +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right). +$$
      @@ -402,7 +411,7 @@ $$
    8034. 55
    8035. 56
    8036. ...
    8037. -
    8038. 66
    8039. +
    8040. 62
    8041. »
    8042. diff --git a/doc/pub/week34/html/._week34-bs047.html b/doc/pub/week34/html/._week34-bs047.html index 21fbe57a9..445497db5 100644 --- a/doc/pub/week34/html/._week34-bs047.html +++ b/doc/pub/week34/html/._week34-bs047.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8043. Installing R, C++, cython or Julia
    8044. Installing R, C++, cython, Numba etc
    8045. Numpy examples and Important Matrix and vector handling packages
    8046. -
    8047. Basic Matrix Features
    8048. -
    8049.    Some famous Matrices
    8050. -
    8051.    More Basic Matrix Features
    8052. -
    8053. Numpy and arrays
    8054. -
    8055. Matrices in Python
    8056. -
    8057. Meet the Pandas
    8058. -
    8059. Friday August 27
    8060. -
    8061.    Simple linear regression model using scikit-learn
    8062. -
    8063.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8064. -
    8065.    Organizing our data
    8066. -
    8067.    Seeing the wood for the trees
    8068. -
    8069.    And what about using neural networks?
    8070. -
    8071. A first summary
    8072. -
    8073. Why Linear Regression (aka Ordinary Least Squares and family)
    8074. -
    8075. Regression analysis, overarching aims
    8076. -
    8077. Regression analysis, overarching aims II
    8078. -
    8079. Examples
    8080. -
    8081. General linear models
    8082. -
    8083. Rewriting the fitting procedure as a linear algebra problem
    8084. -
    8085. Rewriting the fitting procedure as a linear algebra problem, more details
    8086. -
    8087. Generalizing the fitting procedure as a linear algebra problem
    8088. -
    8089. Generalizing the fitting procedure as a linear algebra problem
    8090. -
    8091. Optimizing our parameters
    8092. -
    8093. Our model for the nuclear binding energies
    8094. -
    8095. Optimizing our parameters, more details
    8096. -
    8097. Interpretations and optimizing our parameters
    8098. -
    8099. Interpretations and optimizing our parameters
    8100. -
    8101. Some useful matrix and vector expressions
    8102. -
    8103. Interpretations and optimizing our parameters
    8104. -
    8105. Own code for Ordinary Least Squares
    8106. -
    8107. Adding error analysis and training set up
    8108. -
    8109. The \( \chi^2 \) function
    8110. -
    8111. The \( \chi^2 \) function
    8112. -
    8113. The \( \chi^2 \) function
    8114. -
    8115. The \( \chi^2 \) function
    8116. -
    8117. The \( \chi^2 \) function
    8118. -
    8119. The \( \chi^2 \) function
    8120. -
    8121. Fitting an Equation of State for Dense Nuclear Matter
    8122. -
    8123. The code
    8124. -
    8125. Splitting our Data in Training and Test data
    8126. -
    8127. Exercises
    8128. -
    8129. Exercise 1: Setting up various Python environments
    8130. -
    8131. Exercise 2: making your own data and exploring scikit-learn
    8132. -
    8133. Exercise 3: Normalizing our data
    8134. +
    8135. Numpy and arrays
    8136. +
    8137. Matrices in Python
    8138. +
    8139. Meet the Pandas
    8140. +
    8141.    Simple linear regression model using scikit-learn
    8142. +
    8143.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8144. +
    8145.    Organizing our data
    8146. +
    8147.    And what about using neural networks?
    8148. +
    8149. A first summary
    8150. +
    8151. Why Linear Regression (aka Ordinary Least Squares and family)
    8152. +
    8153. Regression analysis, overarching aims
    8154. +
    8155. Regression analysis, overarching aims II
    8156. +
    8157. Examples
    8158. +
    8159. General linear models
    8160. +
    8161. Rewriting the fitting procedure as a linear algebra problem
    8162. +
    8163. Rewriting the fitting procedure as a linear algebra problem, more details
    8164. +
    8165. Generalizing the fitting procedure as a linear algebra problem
    8166. +
    8167. Generalizing the fitting procedure as a linear algebra problem
    8168. +
    8169. Optimizing our parameters
    8170. +
    8171. Our model for the nuclear binding energies
    8172. +
    8173. Optimizing our parameters, more details
    8174. +
    8175. Interpretations and optimizing our parameters
    8176. +
    8177. Interpretations and optimizing our parameters
    8178. +
    8179. Some useful matrix and vector expressions
    8180. +
    8181. Interpretations and optimizing our parameters
    8182. +
    8183. Own code for Ordinary Least Squares
    8184. +
    8185. Adding error analysis and training set up
    8186. +
    8187. The \( \chi^2 \) function
    8188. +
    8189. The \( \chi^2 \) function
    8190. +
    8191. The \( \chi^2 \) function
    8192. +
    8193. The \( \chi^2 \) function
    8194. +
    8195. The \( \chi^2 \) function
    8196. +
    8197. The \( \chi^2 \) function
    8198. +
    8199. Fitting an Equation of State for Dense Nuclear Matter
    8200. +
    8201. The code
    8202. +
    8203. Splitting our Data in Training and Test data
    8204. +
    8205. Exercises
    8206. +
    8207. Exercise 1: Setting up various Python environments
    8208. +
    8209. Exercise 2: making your own data and exploring scikit-learn
    8210. +
    8211. Exercise 3: Split data in test and training data
    8212. @@ -351,31 +335,48 @@ MathJax.Hub.Config({

       

       

       

      -

      Optimizing our parameters

      +

      Interpretations and optimizing our parameters

      -

      We have defined the matrix \( \boldsymbol{X} \) via the equations

      +

      We can rewrite

      $$ -\begin{align*} -y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ -y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ -y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_1\\ -\dots & \dots \\ -y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_1\\ -\dots & \dots \\ -y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ -\end{align*} +\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right), $$ -

      As we noted above, we stayed with a system with the design matrix - \( \boldsymbol{X}\in {\mathbb{R}}^{n\times n} \), that is we have \( p=n \). For reasons to come later (algorithmic arguments) we will hereafter define -our matrix as \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \), with the predictors refering to the column numbers and the entries \( n \) being the row elements. +

      as

      +$$ +\boldsymbol{X}^T\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{X}\boldsymbol{\beta}, +$$ + +

      and if the matrix \( \boldsymbol{X}^T\boldsymbol{X} \) is invertible we have the solution

      +$$ +\boldsymbol{\beta} =\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}. +$$ + +

      We note also that since our design matrix is defined as \( \boldsymbol{X}\in +{\mathbb{R}}^{n\times p} \), the product \( \boldsymbol{X}^T\boldsymbol{X} \in +{\mathbb{R}}^{p\times p} \). In the above case we have that \( p \ll n \), +in our case \( p=5 \) meaning that we end up with inverting a small +\( 5\times 5 \) matrix. This is a rather common situation, in many cases we end up with low-dimensional +matrices to invert. The methods discussed here and for many other +supervised learning algorithms like classification with logistic +regression or support vector machines, exhibit dimensionalities which +allow for the usage of direct linear algebra methods such as LU decomposition or Singular Value Decomposition (SVD) for finding the inverse of the matrix +\( \boldsymbol{X}^T\boldsymbol{X} \).

      +
      +
      + +

      Small question: Do you think the example we have at hand here (the nuclear binding energies) can lead to problems in inverting the matrix \( \boldsymbol{X}^T\boldsymbol{X} \)? What kind of problems can we expect?

      +
      +
      + +

        @@ -401,7 +402,7 @@ our matrix as \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \), with the predict
      • 56
      • 57
      • ...
      • -
      • 66
      • +
      • 62
      • »
      diff --git a/doc/pub/week34/html/._week34-bs048.html b/doc/pub/week34/html/._week34-bs048.html index 39e23ceb8..1b95dc910 100644 --- a/doc/pub/week34/html/._week34-bs048.html +++ b/doc/pub/week34/html/._week34-bs048.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8213. Installing R, C++, cython or Julia
    8214. Installing R, C++, cython, Numba etc
    8215. Numpy examples and Important Matrix and vector handling packages
    8216. -
    8217. Basic Matrix Features
    8218. -
    8219.    Some famous Matrices
    8220. -
    8221.    More Basic Matrix Features
    8222. -
    8223. Numpy and arrays
    8224. -
    8225. Matrices in Python
    8226. -
    8227. Meet the Pandas
    8228. -
    8229. Friday August 27
    8230. -
    8231.    Simple linear regression model using scikit-learn
    8232. -
    8233.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8234. -
    8235.    Organizing our data
    8236. -
    8237.    Seeing the wood for the trees
    8238. -
    8239.    And what about using neural networks?
    8240. -
    8241. A first summary
    8242. -
    8243. Why Linear Regression (aka Ordinary Least Squares and family)
    8244. -
    8245. Regression analysis, overarching aims
    8246. -
    8247. Regression analysis, overarching aims II
    8248. -
    8249. Examples
    8250. -
    8251. General linear models
    8252. -
    8253. Rewriting the fitting procedure as a linear algebra problem
    8254. -
    8255. Rewriting the fitting procedure as a linear algebra problem, more details
    8256. -
    8257. Generalizing the fitting procedure as a linear algebra problem
    8258. -
    8259. Generalizing the fitting procedure as a linear algebra problem
    8260. -
    8261. Optimizing our parameters
    8262. -
    8263. Our model for the nuclear binding energies
    8264. -
    8265. Optimizing our parameters, more details
    8266. -
    8267. Interpretations and optimizing our parameters
    8268. -
    8269. Interpretations and optimizing our parameters
    8270. -
    8271. Some useful matrix and vector expressions
    8272. -
    8273. Interpretations and optimizing our parameters
    8274. -
    8275. Own code for Ordinary Least Squares
    8276. -
    8277. Adding error analysis and training set up
    8278. -
    8279. The \( \chi^2 \) function
    8280. -
    8281. The \( \chi^2 \) function
    8282. -
    8283. The \( \chi^2 \) function
    8284. -
    8285. The \( \chi^2 \) function
    8286. -
    8287. The \( \chi^2 \) function
    8288. -
    8289. The \( \chi^2 \) function
    8290. -
    8291. Fitting an Equation of State for Dense Nuclear Matter
    8292. -
    8293. The code
    8294. -
    8295. Splitting our Data in Training and Test data
    8296. -
    8297. Exercises
    8298. -
    8299. Exercise 1: Setting up various Python environments
    8300. -
    8301. Exercise 2: making your own data and exploring scikit-learn
    8302. -
    8303. Exercise 3: Normalizing our data
    8304. +
    8305. Numpy and arrays
    8306. +
    8307. Matrices in Python
    8308. +
    8309. Meet the Pandas
    8310. +
    8311.    Simple linear regression model using scikit-learn
    8312. +
    8313.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8314. +
    8315.    Organizing our data
    8316. +
    8317.    And what about using neural networks?
    8318. +
    8319. A first summary
    8320. +
    8321. Why Linear Regression (aka Ordinary Least Squares and family)
    8322. +
    8323. Regression analysis, overarching aims
    8324. +
    8325. Regression analysis, overarching aims II
    8326. +
    8327. Examples
    8328. +
    8329. General linear models
    8330. +
    8331. Rewriting the fitting procedure as a linear algebra problem
    8332. +
    8333. Rewriting the fitting procedure as a linear algebra problem, more details
    8334. +
    8335. Generalizing the fitting procedure as a linear algebra problem
    8336. +
    8337. Generalizing the fitting procedure as a linear algebra problem
    8338. +
    8339. Optimizing our parameters
    8340. +
    8341. Our model for the nuclear binding energies
    8342. +
    8343. Optimizing our parameters, more details
    8344. +
    8345. Interpretations and optimizing our parameters
    8346. +
    8347. Interpretations and optimizing our parameters
    8348. +
    8349. Some useful matrix and vector expressions
    8350. +
    8351. Interpretations and optimizing our parameters
    8352. +
    8353. Own code for Ordinary Least Squares
    8354. +
    8355. Adding error analysis and training set up
    8356. +
    8357. The \( \chi^2 \) function
    8358. +
    8359. The \( \chi^2 \) function
    8360. +
    8361. The \( \chi^2 \) function
    8362. +
    8363. The \( \chi^2 \) function
    8364. +
    8365. The \( \chi^2 \) function
    8366. +
    8367. The \( \chi^2 \) function
    8368. +
    8369. Fitting an Equation of State for Dense Nuclear Matter
    8370. +
    8371. The code
    8372. +
    8373. Splitting our Data in Training and Test data
    8374. +
    8375. Exercises
    8376. +
    8377. Exercise 1: Setting up various Python environments
    8378. +
    8379. Exercise 2: making your own data and exploring scikit-learn
    8380. +
    8381. Exercise 3: Split data in test and training data
    8382. @@ -351,108 +335,11 @@ MathJax.Hub.Config({

       

       

       

      -

      Our model for the nuclear binding energies

      +

      Some useful matrix and vector expressions

      -

      In our introductory notes we looked at the so-called liquid drop model. Let us remind ourselves about what we did by looking at the code.

      +

      See the handwritten notes at https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/2022/NotesExercise5Week452022.pdf

      -

      We restate the parts of the code we are most interested in.

      - - -
      -
      -
      -
      -
      -
      # Common imports
      -import numpy as np
      -import pandas as pd
      -import matplotlib.pyplot as plt
      -from IPython.display import display
      -import os
      -
      -# Where to save the figures and data files
      -PROJECT_ROOT_DIR = "Results"
      -FIGURE_ID = "Results/FigureFiles"
      -DATA_ID = "DataFiles/"
      -
      -if not os.path.exists(PROJECT_ROOT_DIR):
      -    os.mkdir(PROJECT_ROOT_DIR)
      -
      -if not os.path.exists(FIGURE_ID):
      -    os.makedirs(FIGURE_ID)
      -
      -if not os.path.exists(DATA_ID):
      -    os.makedirs(DATA_ID)
      -
      -def image_path(fig_id):
      -    return os.path.join(FIGURE_ID, fig_id)
      -
      -def data_path(dat_id):
      -    return os.path.join(DATA_ID, dat_id)
      -
      -def save_fig(fig_id):
      -    plt.savefig(image_path(fig_id) + ".png", format='png')
      -
      -infile = open(data_path("MassEval2016.dat"),'r')
      -
      -
      -# Read the experimental data with Pandas
      -Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),
      -              names=('N', 'Z', 'A', 'Element', 'Ebinding'),
      -              widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),
      -              header=39,
      -              index_col=False)
      -
      -# Extrapolated values are indicated by '#' in place of the decimal place, so
      -# the Ebinding column won't be numeric. Coerce to float and drop these entries.
      -Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')
      -Masses = Masses.dropna()
      -# Convert from keV to MeV.
      -Masses['Ebinding'] /= 1000
      -
      -# Group the DataFrame by nucleon number, A.
      -Masses = Masses.groupby('A')
      -# Find the rows of the grouped DataFrame with the maximum binding energy.
      -Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])
      -A = Masses['A']
      -Z = Masses['Z']
      -N = Masses['N']
      -Element = Masses['Element']
      -Energies = Masses['Ebinding']
      -
      -# Now we set up the design matrix X
      -X = np.zeros((len(A),5))
      -X[:,0] = 1
      -X[:,1] = A
      -X[:,2] = A**(2.0/3.0)
      -X[:,3] = A**(-1.0/3.0)
      -X[:,4] = A**(-1.0)
      -# Then nice printout using pandas
      -DesignMatrix = pd.DataFrame(X)
      -DesignMatrix.index = A
      -DesignMatrix.columns = ['1', 'A', 'A^(2/3)', 'A^(-1/3)', '1/A']
      -display(DesignMatrix)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      With \( \boldsymbol{\beta}\in {\mathbb{R}}^{p\times 1} \), it means that we will hereafter write our equations for the approximation as

      -$$ -\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, -$$ - -

      throughout these lectures.

      +

      These notes will be discussed during one of the lectures.

      @@ -479,7 +366,7 @@ $$

    8383. 57
    8384. 58
    8385. ...
    8386. -
    8387. 66
    8388. +
    8389. 62
    8390. »
    8391. diff --git a/doc/pub/week34/html/._week34-bs049.html b/doc/pub/week34/html/._week34-bs049.html index 2133bf74e..ff24777d5 100644 --- a/doc/pub/week34/html/._week34-bs049.html +++ b/doc/pub/week34/html/._week34-bs049.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8392. Installing R, C++, cython or Julia
    8393. Installing R, C++, cython, Numba etc
    8394. Numpy examples and Important Matrix and vector handling packages
    8395. -
    8396. Basic Matrix Features
    8397. -
    8398.    Some famous Matrices
    8399. -
    8400.    More Basic Matrix Features
    8401. -
    8402. Numpy and arrays
    8403. -
    8404. Matrices in Python
    8405. -
    8406. Meet the Pandas
    8407. -
    8408. Friday August 27
    8409. -
    8410.    Simple linear regression model using scikit-learn
    8411. -
    8412.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8413. -
    8414.    Organizing our data
    8415. -
    8416.    Seeing the wood for the trees
    8417. -
    8418.    And what about using neural networks?
    8419. -
    8420. A first summary
    8421. -
    8422. Why Linear Regression (aka Ordinary Least Squares and family)
    8423. -
    8424. Regression analysis, overarching aims
    8425. -
    8426. Regression analysis, overarching aims II
    8427. -
    8428. Examples
    8429. -
    8430. General linear models
    8431. -
    8432. Rewriting the fitting procedure as a linear algebra problem
    8433. -
    8434. Rewriting the fitting procedure as a linear algebra problem, more details
    8435. -
    8436. Generalizing the fitting procedure as a linear algebra problem
    8437. -
    8438. Generalizing the fitting procedure as a linear algebra problem
    8439. -
    8440. Optimizing our parameters
    8441. -
    8442. Our model for the nuclear binding energies
    8443. -
    8444. Optimizing our parameters, more details
    8445. -
    8446. Interpretations and optimizing our parameters
    8447. -
    8448. Interpretations and optimizing our parameters
    8449. -
    8450. Some useful matrix and vector expressions
    8451. -
    8452. Interpretations and optimizing our parameters
    8453. -
    8454. Own code for Ordinary Least Squares
    8455. -
    8456. Adding error analysis and training set up
    8457. -
    8458. The \( \chi^2 \) function
    8459. -
    8460. The \( \chi^2 \) function
    8461. -
    8462. The \( \chi^2 \) function
    8463. -
    8464. The \( \chi^2 \) function
    8465. -
    8466. The \( \chi^2 \) function
    8467. -
    8468. The \( \chi^2 \) function
    8469. -
    8470. Fitting an Equation of State for Dense Nuclear Matter
    8471. -
    8472. The code
    8473. -
    8474. Splitting our Data in Training and Test data
    8475. -
    8476. Exercises
    8477. -
    8478. Exercise 1: Setting up various Python environments
    8479. -
    8480. Exercise 2: making your own data and exploring scikit-learn
    8481. -
    8482. Exercise 3: Normalizing our data
    8483. +
    8484. Numpy and arrays
    8485. +
    8486. Matrices in Python
    8487. +
    8488. Meet the Pandas
    8489. +
    8490.    Simple linear regression model using scikit-learn
    8491. +
    8492.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8493. +
    8494.    Organizing our data
    8495. +
    8496.    And what about using neural networks?
    8497. +
    8498. A first summary
    8499. +
    8500. Why Linear Regression (aka Ordinary Least Squares and family)
    8501. +
    8502. Regression analysis, overarching aims
    8503. +
    8504. Regression analysis, overarching aims II
    8505. +
    8506. Examples
    8507. +
    8508. General linear models
    8509. +
    8510. Rewriting the fitting procedure as a linear algebra problem
    8511. +
    8512. Rewriting the fitting procedure as a linear algebra problem, more details
    8513. +
    8514. Generalizing the fitting procedure as a linear algebra problem
    8515. +
    8516. Generalizing the fitting procedure as a linear algebra problem
    8517. +
    8518. Optimizing our parameters
    8519. +
    8520. Our model for the nuclear binding energies
    8521. +
    8522. Optimizing our parameters, more details
    8523. +
    8524. Interpretations and optimizing our parameters
    8525. +
    8526. Interpretations and optimizing our parameters
    8527. +
    8528. Some useful matrix and vector expressions
    8529. +
    8530. Interpretations and optimizing our parameters
    8531. +
    8532. Own code for Ordinary Least Squares
    8533. +
    8534. Adding error analysis and training set up
    8535. +
    8536. The \( \chi^2 \) function
    8537. +
    8538. The \( \chi^2 \) function
    8539. +
    8540. The \( \chi^2 \) function
    8541. +
    8542. The \( \chi^2 \) function
    8543. +
    8544. The \( \chi^2 \) function
    8545. +
    8546. The \( \chi^2 \) function
    8547. +
    8548. Fitting an Equation of State for Dense Nuclear Matter
    8549. +
    8550. The code
    8551. +
    8552. Splitting our Data in Training and Test data
    8553. +
    8554. Exercises
    8555. +
    8556. Exercise 1: Setting up various Python environments
    8557. +
    8558. Exercise 2: making your own data and exploring scikit-learn
    8559. +
    8560. Exercise 3: Split data in test and training data
    8561. @@ -351,40 +335,32 @@ MathJax.Hub.Config({

       

       

       

      -

      Optimizing our parameters, more details

      +

      Interpretations and optimizing our parameters

      -

      With the above we use the design matrix to define the approximation \( \boldsymbol{\tilde{y}} \) via the unknown quantity \( \boldsymbol{\beta} \) as

      +

      The residuals \( \boldsymbol{\epsilon} \) are in turn given by

      $$ -\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +\boldsymbol{\epsilon} = \boldsymbol{y}-\boldsymbol{\tilde{y}} = \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}, $$ -

      and in order to find the optimal parameters \( \beta_i \) instead of solving the above linear algebra problem, we define a function which gives a measure of the spread between the values \( y_i \) (which represent hopefully the exact values) and the parameterized values \( \tilde{y}_i \), namely

      +

      and with

      $$ -C(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, $$ -

      or using the matrix \( \boldsymbol{X} \) and in a more compact matrix-vector notation as

      +

      we have

      $$ -C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +\boldsymbol{X}^T\boldsymbol{\epsilon}=\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, $$ -

      This function is one possible way to define the so-called cost function.

      - -

      It is also common to define -the function \( C \) as -

      - -$$ -C(\boldsymbol{\beta})=\frac{1}{2n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2, -$$ - -

      since when taking the first derivative with respect to the unknown parameters \( \beta \), the factor of \( 2 \) cancels out.

      +

      meaning that the solution for \( \boldsymbol{\beta} \) is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach.

      +

      Let us now return to our nuclear binding energies and simply code the above equations.

      +

      diff --git a/doc/pub/week34/html/._week34-bs050.html b/doc/pub/week34/html/._week34-bs050.html index 953f6aa7f..473b48e58 100644 --- a/doc/pub/week34/html/._week34-bs050.html +++ b/doc/pub/week34/html/._week34-bs050.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8562. Installing R, C++, cython or Julia
    8563. Installing R, C++, cython, Numba etc
    8564. Numpy examples and Important Matrix and vector handling packages
    8565. -
    8566. Basic Matrix Features
    8567. -
    8568.    Some famous Matrices
    8569. -
    8570.    More Basic Matrix Features
    8571. -
    8572. Numpy and arrays
    8573. -
    8574. Matrices in Python
    8575. -
    8576. Meet the Pandas
    8577. -
    8578. Friday August 27
    8579. -
    8580.    Simple linear regression model using scikit-learn
    8581. -
    8582.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8583. -
    8584.    Organizing our data
    8585. -
    8586.    Seeing the wood for the trees
    8587. -
    8588.    And what about using neural networks?
    8589. -
    8590. A first summary
    8591. -
    8592. Why Linear Regression (aka Ordinary Least Squares and family)
    8593. -
    8594. Regression analysis, overarching aims
    8595. -
    8596. Regression analysis, overarching aims II
    8597. -
    8598. Examples
    8599. -
    8600. General linear models
    8601. -
    8602. Rewriting the fitting procedure as a linear algebra problem
    8603. -
    8604. Rewriting the fitting procedure as a linear algebra problem, more details
    8605. -
    8606. Generalizing the fitting procedure as a linear algebra problem
    8607. -
    8608. Generalizing the fitting procedure as a linear algebra problem
    8609. -
    8610. Optimizing our parameters
    8611. -
    8612. Our model for the nuclear binding energies
    8613. -
    8614. Optimizing our parameters, more details
    8615. -
    8616. Interpretations and optimizing our parameters
    8617. -
    8618. Interpretations and optimizing our parameters
    8619. -
    8620. Some useful matrix and vector expressions
    8621. -
    8622. Interpretations and optimizing our parameters
    8623. -
    8624. Own code for Ordinary Least Squares
    8625. -
    8626. Adding error analysis and training set up
    8627. -
    8628. The \( \chi^2 \) function
    8629. -
    8630. The \( \chi^2 \) function
    8631. -
    8632. The \( \chi^2 \) function
    8633. -
    8634. The \( \chi^2 \) function
    8635. -
    8636. The \( \chi^2 \) function
    8637. -
    8638. The \( \chi^2 \) function
    8639. -
    8640. Fitting an Equation of State for Dense Nuclear Matter
    8641. -
    8642. The code
    8643. -
    8644. Splitting our Data in Training and Test data
    8645. -
    8646. Exercises
    8647. -
    8648. Exercise 1: Setting up various Python environments
    8649. -
    8650. Exercise 2: making your own data and exploring scikit-learn
    8651. -
    8652. Exercise 3: Normalizing our data
    8653. +
    8654. Numpy and arrays
    8655. +
    8656. Matrices in Python
    8657. +
    8658. Meet the Pandas
    8659. +
    8660.    Simple linear regression model using scikit-learn
    8661. +
    8662.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8663. +
    8664.    Organizing our data
    8665. +
    8666.    And what about using neural networks?
    8667. +
    8668. A first summary
    8669. +
    8670. Why Linear Regression (aka Ordinary Least Squares and family)
    8671. +
    8672. Regression analysis, overarching aims
    8673. +
    8674. Regression analysis, overarching aims II
    8675. +
    8676. Examples
    8677. +
    8678. General linear models
    8679. +
    8680. Rewriting the fitting procedure as a linear algebra problem
    8681. +
    8682. Rewriting the fitting procedure as a linear algebra problem, more details
    8683. +
    8684. Generalizing the fitting procedure as a linear algebra problem
    8685. +
    8686. Generalizing the fitting procedure as a linear algebra problem
    8687. +
    8688. Optimizing our parameters
    8689. +
    8690. Our model for the nuclear binding energies
    8691. +
    8692. Optimizing our parameters, more details
    8693. +
    8694. Interpretations and optimizing our parameters
    8695. +
    8696. Interpretations and optimizing our parameters
    8697. +
    8698. Some useful matrix and vector expressions
    8699. +
    8700. Interpretations and optimizing our parameters
    8701. +
    8702. Own code for Ordinary Least Squares
    8703. +
    8704. Adding error analysis and training set up
    8705. +
    8706. The \( \chi^2 \) function
    8707. +
    8708. The \( \chi^2 \) function
    8709. +
    8710. The \( \chi^2 \) function
    8711. +
    8712. The \( \chi^2 \) function
    8713. +
    8714. The \( \chi^2 \) function
    8715. +
    8716. The \( \chi^2 \) function
    8717. +
    8718. Fitting an Equation of State for Dense Nuclear Matter
    8719. +
    8720. The code
    8721. +
    8722. Splitting our Data in Training and Test data
    8723. +
    8724. Exercises
    8725. +
    8726. Exercise 1: Setting up various Python environments
    8727. +
    8728. Exercise 2: making your own data and exploring scikit-learn
    8729. +
    8730. Exercise 3: Split data in test and training data
    8731. @@ -351,54 +335,95 @@ MathJax.Hub.Config({

       

       

       

      -

      Interpretations and optimizing our parameters

      -
      -
      - +

      Own code for Ordinary Least Squares

      -

      The function

      -$$ -C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}, -$$ - -

      can be linked to the variance of the quantity \( y_i \) if we interpret the latter as the mean value. -When linking (see the discussion below) with the maximum likelihood approach below, we will indeed interpret \( y_i \) as a mean value -

      -$$ -y_{i}=\langle y_i \rangle = \beta_0x_{i,0}+\beta_1x_{i,1}+\beta_2x_{i,2}+\dots+\beta_{n-1}x_{i,n-1}+\epsilon_i, -$$ - -

      where \( \langle y_i \rangle \) is the mean value. Keep in mind also that -till now we have treated \( y_i \) as the exact value. Normally, the -response (dependent or outcome) variable \( y_i \) the outcome of a -numerical experiment or another type of experiment and is thus only an -approximation to the true value. It is then always accompanied by an -error estimate, often limited to a statistical error estimate given by -the standard deviation discussed earlier. In the discussion here we -will treat \( y_i \) as our exact value for the response variable. +

      It is rather straightforward to implement the matrix inversion and obtain the parameters \( \boldsymbol{\beta} \). After having defined the matrix \( \boldsymbol{X} \) we simply need to +write

      -

      In order to find the parameters \( \beta_i \) we will then minimize the spread of \( C(\boldsymbol{\beta}) \), that is we are going to solve the problem

      -$$ -{\displaystyle \min_{\boldsymbol{\beta}\in -{\mathbb{R}}^{p}}}\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. -$$ - -

      In practical terms it means we will require

      -$$ -\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)^2\right]=0, -$$ - -

      which results in

      -$$ -\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_{ij}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)\right]=0, -$$ - -

      or in a matrix-vector form as

      -$$ -\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right). -$$ + +
      +
      +
      +
      +
      +
      # matrix inversion to find beta
      +beta = np.linalg.inv(X.T.dot(X)).dot(X.T).dot(Energies)
      +# and then make the prediction
      +ytilde = X @ beta
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      Alternatively, you can use the least squares functionality in Numpy as

      + + +
      +
      +
      +
      +
      +
      fit = np.linalg.lstsq(X, Energies, rcond =None)[0]
      +ytildenp = np.dot(fit,X.T)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      And finally we plot our fit with and compare with data

      + + +
      +
      +
      +
      +
      +
      Masses['Eapprox']  = ytilde
      +# Generate a plot comparing the experimental with the fitted values values.
      +fig, ax = plt.subplots()
      +ax.set_xlabel(r'$A = N + Z$')
      +ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$')
      +ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,
      +            label='Ame2016')
      +ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',
      +            label='Fit')
      +ax.legend()
      +save_fig("Masses2016OLS")
      +plt.show()
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      @@ -427,7 +452,7 @@ $$
    8732. 59
    8733. 60
    8734. ...
    8735. -
    8736. 66
    8737. +
    8738. 62
    8739. »
    8740. diff --git a/doc/pub/week34/html/._week34-bs051.html b/doc/pub/week34/html/._week34-bs051.html index 0448bc1a8..2a8298132 100644 --- a/doc/pub/week34/html/._week34-bs051.html +++ b/doc/pub/week34/html/._week34-bs051.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8741. Installing R, C++, cython or Julia
    8742. Installing R, C++, cython, Numba etc
    8743. Numpy examples and Important Matrix and vector handling packages
    8744. -
    8745. Basic Matrix Features
    8746. -
    8747.    Some famous Matrices
    8748. -
    8749.    More Basic Matrix Features
    8750. -
    8751. Numpy and arrays
    8752. -
    8753. Matrices in Python
    8754. -
    8755. Meet the Pandas
    8756. -
    8757. Friday August 27
    8758. -
    8759.    Simple linear regression model using scikit-learn
    8760. -
    8761.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8762. -
    8763.    Organizing our data
    8764. -
    8765.    Seeing the wood for the trees
    8766. -
    8767.    And what about using neural networks?
    8768. -
    8769. A first summary
    8770. -
    8771. Why Linear Regression (aka Ordinary Least Squares and family)
    8772. -
    8773. Regression analysis, overarching aims
    8774. -
    8775. Regression analysis, overarching aims II
    8776. -
    8777. Examples
    8778. -
    8779. General linear models
    8780. -
    8781. Rewriting the fitting procedure as a linear algebra problem
    8782. -
    8783. Rewriting the fitting procedure as a linear algebra problem, more details
    8784. -
    8785. Generalizing the fitting procedure as a linear algebra problem
    8786. -
    8787. Generalizing the fitting procedure as a linear algebra problem
    8788. -
    8789. Optimizing our parameters
    8790. -
    8791. Our model for the nuclear binding energies
    8792. -
    8793. Optimizing our parameters, more details
    8794. -
    8795. Interpretations and optimizing our parameters
    8796. -
    8797. Interpretations and optimizing our parameters
    8798. -
    8799. Some useful matrix and vector expressions
    8800. -
    8801. Interpretations and optimizing our parameters
    8802. -
    8803. Own code for Ordinary Least Squares
    8804. -
    8805. Adding error analysis and training set up
    8806. -
    8807. The \( \chi^2 \) function
    8808. -
    8809. The \( \chi^2 \) function
    8810. -
    8811. The \( \chi^2 \) function
    8812. -
    8813. The \( \chi^2 \) function
    8814. -
    8815. The \( \chi^2 \) function
    8816. -
    8817. The \( \chi^2 \) function
    8818. -
    8819. Fitting an Equation of State for Dense Nuclear Matter
    8820. -
    8821. The code
    8822. -
    8823. Splitting our Data in Training and Test data
    8824. -
    8825. Exercises
    8826. -
    8827. Exercise 1: Setting up various Python environments
    8828. -
    8829. Exercise 2: making your own data and exploring scikit-learn
    8830. -
    8831. Exercise 3: Normalizing our data
    8832. +
    8833. Numpy and arrays
    8834. +
    8835. Matrices in Python
    8836. +
    8837. Meet the Pandas
    8838. +
    8839.    Simple linear regression model using scikit-learn
    8840. +
    8841.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8842. +
    8843.    Organizing our data
    8844. +
    8845.    And what about using neural networks?
    8846. +
    8847. A first summary
    8848. +
    8849. Why Linear Regression (aka Ordinary Least Squares and family)
    8850. +
    8851. Regression analysis, overarching aims
    8852. +
    8853. Regression analysis, overarching aims II
    8854. +
    8855. Examples
    8856. +
    8857. General linear models
    8858. +
    8859. Rewriting the fitting procedure as a linear algebra problem
    8860. +
    8861. Rewriting the fitting procedure as a linear algebra problem, more details
    8862. +
    8863. Generalizing the fitting procedure as a linear algebra problem
    8864. +
    8865. Generalizing the fitting procedure as a linear algebra problem
    8866. +
    8867. Optimizing our parameters
    8868. +
    8869. Our model for the nuclear binding energies
    8870. +
    8871. Optimizing our parameters, more details
    8872. +
    8873. Interpretations and optimizing our parameters
    8874. +
    8875. Interpretations and optimizing our parameters
    8876. +
    8877. Some useful matrix and vector expressions
    8878. +
    8879. Interpretations and optimizing our parameters
    8880. +
    8881. Own code for Ordinary Least Squares
    8882. +
    8883. Adding error analysis and training set up
    8884. +
    8885. The \( \chi^2 \) function
    8886. +
    8887. The \( \chi^2 \) function
    8888. +
    8889. The \( \chi^2 \) function
    8890. +
    8891. The \( \chi^2 \) function
    8892. +
    8893. The \( \chi^2 \) function
    8894. +
    8895. The \( \chi^2 \) function
    8896. +
    8897. Fitting an Equation of State for Dense Nuclear Matter
    8898. +
    8899. The code
    8900. +
    8901. Splitting our Data in Training and Test data
    8902. +
    8903. Exercises
    8904. +
    8905. Exercise 1: Setting up various Python environments
    8906. +
    8907. Exercise 2: making your own data and exploring scikit-learn
    8908. +
    8909. Exercise 3: Split data in test and training data
    8910. @@ -351,45 +335,111 @@ MathJax.Hub.Config({

       

       

       

      -

      Interpretations and optimizing our parameters

      -
      -
      - -

      We can rewrite

      -$$ -\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right), -$$ +

      Adding error analysis and training set up

      -

      as

      -$$ -\boldsymbol{X}^T\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{X}\boldsymbol{\beta}, -$$ - -

      and if the matrix \( \boldsymbol{X}^T\boldsymbol{X} \) is invertible we have the solution

      -$$ -\boldsymbol{\beta} =\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}. -$$ - -

      We note also that since our design matrix is defined as \( \boldsymbol{X}\in -{\mathbb{R}}^{n\times p} \), the product \( \boldsymbol{X}^T\boldsymbol{X} \in -{\mathbb{R}}^{p\times p} \). In the above case we have that \( p \ll n \), -in our case \( p=5 \) meaning that we end up with inverting a small -\( 5\times 5 \) matrix. This is a rather common situation, in many cases we end up with low-dimensional -matrices to invert. The methods discussed here and for many other -supervised learning algorithms like classification with logistic -regression or support vector machines, exhibit dimensionalities which -allow for the usage of direct linear algebra methods such as LU decomposition or Singular Value Decomposition (SVD) for finding the inverse of the matrix -\( \boldsymbol{X}^T\boldsymbol{X} \). +

      We can easily test our fit by computing the \( R2 \) score that we discussed in connection with the functionality of Scikit-Learn in the introductory slides. +Since we are not using Scikit-Learn here we can define our own \( R2 \) function as

      + + +
      +
      +
      +
      +
      +
      def R2(y_data, y_model):
      +    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +

      and we would be using it as

      -
      -
      - -

      Small question: Do you think the example we have at hand here (the nuclear binding energies) can lead to problems in inverting the matrix \( \boldsymbol{X}^T\boldsymbol{X} \)? What kind of problems can we expect?

      + +
      +
      +
      +
      +
      +
      print(R2(Energies,ytilde))
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      We can easily add our MSE score as

      + + +
      +
      +
      +
      +
      +
      def MSE(y_data,y_model):
      +    n = np.size(y_model)
      +    return np.sum((y_data-y_model)**2)/n
      +
      +print(MSE(Energies,ytilde))
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      and finally the relative error as

      + + +
      +
      +
      +
      +
      +
      def RelativeError(y_data,y_model):
      +    return abs((y_data-y_model)/y_data)
      +print(RelativeError(Energies, ytilde))
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      @@ -418,7 +468,7 @@ allow for the usage of direct linear algebra methods such as LU decomposi
    8911. 60
    8912. 61
    8913. ...
    8914. -
    8915. 66
    8916. +
    8917. 62
    8918. »
    8919. diff --git a/doc/pub/week34/html/._week34-bs052.html b/doc/pub/week34/html/._week34-bs052.html index 079491290..9df498381 100644 --- a/doc/pub/week34/html/._week34-bs052.html +++ b/doc/pub/week34/html/._week34-bs052.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    8920. Installing R, C++, cython or Julia
    8921. Installing R, C++, cython, Numba etc
    8922. Numpy examples and Important Matrix and vector handling packages
    8923. -
    8924. Basic Matrix Features
    8925. -
    8926.    Some famous Matrices
    8927. -
    8928.    More Basic Matrix Features
    8929. -
    8930. Numpy and arrays
    8931. -
    8932. Matrices in Python
    8933. -
    8934. Meet the Pandas
    8935. -
    8936. Friday August 27
    8937. -
    8938.    Simple linear regression model using scikit-learn
    8939. -
    8940.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    8941. -
    8942.    Organizing our data
    8943. -
    8944.    Seeing the wood for the trees
    8945. -
    8946.    And what about using neural networks?
    8947. -
    8948. A first summary
    8949. -
    8950. Why Linear Regression (aka Ordinary Least Squares and family)
    8951. -
    8952. Regression analysis, overarching aims
    8953. -
    8954. Regression analysis, overarching aims II
    8955. -
    8956. Examples
    8957. -
    8958. General linear models
    8959. -
    8960. Rewriting the fitting procedure as a linear algebra problem
    8961. -
    8962. Rewriting the fitting procedure as a linear algebra problem, more details
    8963. -
    8964. Generalizing the fitting procedure as a linear algebra problem
    8965. -
    8966. Generalizing the fitting procedure as a linear algebra problem
    8967. -
    8968. Optimizing our parameters
    8969. -
    8970. Our model for the nuclear binding energies
    8971. -
    8972. Optimizing our parameters, more details
    8973. -
    8974. Interpretations and optimizing our parameters
    8975. -
    8976. Interpretations and optimizing our parameters
    8977. -
    8978. Some useful matrix and vector expressions
    8979. -
    8980. Interpretations and optimizing our parameters
    8981. -
    8982. Own code for Ordinary Least Squares
    8983. -
    8984. Adding error analysis and training set up
    8985. -
    8986. The \( \chi^2 \) function
    8987. -
    8988. The \( \chi^2 \) function
    8989. -
    8990. The \( \chi^2 \) function
    8991. -
    8992. The \( \chi^2 \) function
    8993. -
    8994. The \( \chi^2 \) function
    8995. -
    8996. The \( \chi^2 \) function
    8997. -
    8998. Fitting an Equation of State for Dense Nuclear Matter
    8999. -
    9000. The code
    9001. -
    9002. Splitting our Data in Training and Test data
    9003. -
    9004. Exercises
    9005. -
    9006. Exercise 1: Setting up various Python environments
    9007. -
    9008. Exercise 2: making your own data and exploring scikit-learn
    9009. -
    9010. Exercise 3: Normalizing our data
    9011. +
    9012. Numpy and arrays
    9013. +
    9014. Matrices in Python
    9015. +
    9016. Meet the Pandas
    9017. +
    9018.    Simple linear regression model using scikit-learn
    9019. +
    9020.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9021. +
    9022.    Organizing our data
    9023. +
    9024.    And what about using neural networks?
    9025. +
    9026. A first summary
    9027. +
    9028. Why Linear Regression (aka Ordinary Least Squares and family)
    9029. +
    9030. Regression analysis, overarching aims
    9031. +
    9032. Regression analysis, overarching aims II
    9033. +
    9034. Examples
    9035. +
    9036. General linear models
    9037. +
    9038. Rewriting the fitting procedure as a linear algebra problem
    9039. +
    9040. Rewriting the fitting procedure as a linear algebra problem, more details
    9041. +
    9042. Generalizing the fitting procedure as a linear algebra problem
    9043. +
    9044. Generalizing the fitting procedure as a linear algebra problem
    9045. +
    9046. Optimizing our parameters
    9047. +
    9048. Our model for the nuclear binding energies
    9049. +
    9050. Optimizing our parameters, more details
    9051. +
    9052. Interpretations and optimizing our parameters
    9053. +
    9054. Interpretations and optimizing our parameters
    9055. +
    9056. Some useful matrix and vector expressions
    9057. +
    9058. Interpretations and optimizing our parameters
    9059. +
    9060. Own code for Ordinary Least Squares
    9061. +
    9062. Adding error analysis and training set up
    9063. +
    9064. The \( \chi^2 \) function
    9065. +
    9066. The \( \chi^2 \) function
    9067. +
    9068. The \( \chi^2 \) function
    9069. +
    9070. The \( \chi^2 \) function
    9071. +
    9072. The \( \chi^2 \) function
    9073. +
    9074. The \( \chi^2 \) function
    9075. +
    9076. Fitting an Equation of State for Dense Nuclear Matter
    9077. +
    9078. The code
    9079. +
    9080. Splitting our Data in Training and Test data
    9081. +
    9082. Exercises
    9083. +
    9084. Exercise 1: Setting up various Python environments
    9085. +
    9086. Exercise 2: making your own data and exploring scikit-learn
    9087. +
    9088. Exercise 3: Split data in test and training data
    9089. @@ -351,27 +335,33 @@ MathJax.Hub.Config({

       

       

       

      -

      Some useful matrix and vector expressions

      +

      The \( \chi^2 \) function

      +
      +
      + -

      The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and -matrices as upper case boldfaced letters. +

      Normally, the response (dependent or outcome) variable \( y_i \) is the +outcome of a numerical experiment or another type of experiment and is +thus only an approximation to the true value. It is then always +accompanied by an error estimate, often limited to a statistical error +estimate given by the standard deviation discussed earlier. In the +discussion here we will treat \( y_i \) as our exact value for the +response variable. +

      + +

      Introducing the standard deviation \( \sigma_i \) for each measurement +\( y_i \), we define now the \( \chi^2 \) function (omitting the \( 1/n \) term) +as

      $$ -\frac{\partial (\boldsymbol{b}^T\boldsymbol{a})}{\partial \boldsymbol{a}} = \boldsymbol{b}, +\chi^2(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\frac{\left(y_i-\tilde{y}_i\right)^2}{\sigma_i^2}=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\frac{1}{\boldsymbol{\Sigma^2}}\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, $$ -$$ -\frac{\partial (\boldsymbol{a}^T\boldsymbol{A}\boldsymbol{a})}{\partial \boldsymbol{a}} = (\boldsymbol{A}+\boldsymbol{A}^T)\boldsymbol{a}, -$$ +

      where the matrix \( \boldsymbol{\Sigma} \) is a diagonal matrix with \( \sigma_i \) as matrix elements.

      +
      +
      -$$ -\frac{\partial tr(\boldsymbol{B}\boldsymbol{A})}{\partial \boldsymbol{A}} = \boldsymbol{B}^T, -$$ - -$$ -\frac{\partial \log{\vert\boldsymbol{A}\vert}}{\partial \boldsymbol{A}} = (\boldsymbol{A}^{-1})^T. -$$

      @@ -397,8 +387,6 @@ $$

    9090. 60
    9091. 61
    9092. 62
    9093. -
    9094. ...
    9095. -
    9096. 66
    9097. »
    9098. diff --git a/doc/pub/week34/html/._week34-bs053.html b/doc/pub/week34/html/._week34-bs053.html index a3703eeef..ff0386dc0 100644 --- a/doc/pub/week34/html/._week34-bs053.html +++ b/doc/pub/week34/html/._week34-bs053.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    9099. Installing R, C++, cython or Julia
    9100. Installing R, C++, cython, Numba etc
    9101. Numpy examples and Important Matrix and vector handling packages
    9102. -
    9103. Basic Matrix Features
    9104. -
    9105.    Some famous Matrices
    9106. -
    9107.    More Basic Matrix Features
    9108. -
    9109. Numpy and arrays
    9110. -
    9111. Matrices in Python
    9112. -
    9113. Meet the Pandas
    9114. -
    9115. Friday August 27
    9116. -
    9117.    Simple linear regression model using scikit-learn
    9118. -
    9119.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9120. -
    9121.    Organizing our data
    9122. -
    9123.    Seeing the wood for the trees
    9124. -
    9125.    And what about using neural networks?
    9126. -
    9127. A first summary
    9128. -
    9129. Why Linear Regression (aka Ordinary Least Squares and family)
    9130. -
    9131. Regression analysis, overarching aims
    9132. -
    9133. Regression analysis, overarching aims II
    9134. -
    9135. Examples
    9136. -
    9137. General linear models
    9138. -
    9139. Rewriting the fitting procedure as a linear algebra problem
    9140. -
    9141. Rewriting the fitting procedure as a linear algebra problem, more details
    9142. -
    9143. Generalizing the fitting procedure as a linear algebra problem
    9144. -
    9145. Generalizing the fitting procedure as a linear algebra problem
    9146. -
    9147. Optimizing our parameters
    9148. -
    9149. Our model for the nuclear binding energies
    9150. -
    9151. Optimizing our parameters, more details
    9152. -
    9153. Interpretations and optimizing our parameters
    9154. -
    9155. Interpretations and optimizing our parameters
    9156. -
    9157. Some useful matrix and vector expressions
    9158. -
    9159. Interpretations and optimizing our parameters
    9160. -
    9161. Own code for Ordinary Least Squares
    9162. -
    9163. Adding error analysis and training set up
    9164. -
    9165. The \( \chi^2 \) function
    9166. -
    9167. The \( \chi^2 \) function
    9168. -
    9169. The \( \chi^2 \) function
    9170. -
    9171. The \( \chi^2 \) function
    9172. -
    9173. The \( \chi^2 \) function
    9174. -
    9175. The \( \chi^2 \) function
    9176. -
    9177. Fitting an Equation of State for Dense Nuclear Matter
    9178. -
    9179. The code
    9180. -
    9181. Splitting our Data in Training and Test data
    9182. -
    9183. Exercises
    9184. -
    9185. Exercise 1: Setting up various Python environments
    9186. -
    9187. Exercise 2: making your own data and exploring scikit-learn
    9188. -
    9189. Exercise 3: Normalizing our data
    9190. +
    9191. Numpy and arrays
    9192. +
    9193. Matrices in Python
    9194. +
    9195. Meet the Pandas
    9196. +
    9197.    Simple linear regression model using scikit-learn
    9198. +
    9199.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9200. +
    9201.    Organizing our data
    9202. +
    9203.    And what about using neural networks?
    9204. +
    9205. A first summary
    9206. +
    9207. Why Linear Regression (aka Ordinary Least Squares and family)
    9208. +
    9209. Regression analysis, overarching aims
    9210. +
    9211. Regression analysis, overarching aims II
    9212. +
    9213. Examples
    9214. +
    9215. General linear models
    9216. +
    9217. Rewriting the fitting procedure as a linear algebra problem
    9218. +
    9219. Rewriting the fitting procedure as a linear algebra problem, more details
    9220. +
    9221. Generalizing the fitting procedure as a linear algebra problem
    9222. +
    9223. Generalizing the fitting procedure as a linear algebra problem
    9224. +
    9225. Optimizing our parameters
    9226. +
    9227. Our model for the nuclear binding energies
    9228. +
    9229. Optimizing our parameters, more details
    9230. +
    9231. Interpretations and optimizing our parameters
    9232. +
    9233. Interpretations and optimizing our parameters
    9234. +
    9235. Some useful matrix and vector expressions
    9236. +
    9237. Interpretations and optimizing our parameters
    9238. +
    9239. Own code for Ordinary Least Squares
    9240. +
    9241. Adding error analysis and training set up
    9242. +
    9243. The \( \chi^2 \) function
    9244. +
    9245. The \( \chi^2 \) function
    9246. +
    9247. The \( \chi^2 \) function
    9248. +
    9249. The \( \chi^2 \) function
    9250. +
    9251. The \( \chi^2 \) function
    9252. +
    9253. The \( \chi^2 \) function
    9254. +
    9255. Fitting an Equation of State for Dense Nuclear Matter
    9256. +
    9257. The code
    9258. +
    9259. Splitting our Data in Training and Test data
    9260. +
    9261. Exercises
    9262. +
    9263. Exercise 1: Setting up various Python environments
    9264. +
    9265. Exercise 2: making your own data and exploring scikit-learn
    9266. +
    9267. Exercise 3: Split data in test and training data
    9268. @@ -351,32 +335,31 @@ MathJax.Hub.Config({

       

       

       

      -

      Interpretations and optimizing our parameters

      +

      The \( \chi^2 \) function

      -

      The residuals \( \boldsymbol{\epsilon} \) are in turn given by

      + +

      In order to find the parameters \( \beta_i \) we will then minimize the spread of \( \chi^2(\boldsymbol{\beta}) \) by requiring

      $$ -\boldsymbol{\epsilon} = \boldsymbol{y}-\boldsymbol{\tilde{y}} = \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}, +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)^2\right]=0, $$ -

      and with

      +

      which results in

      $$ -\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}\frac{x_{ij}}{\sigma_i}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)\right]=0, $$ -

      we have

      +

      or in a matrix-vector form as

      $$ -\boldsymbol{X}^T\boldsymbol{\epsilon}=\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right). $$ -

      meaning that the solution for \( \boldsymbol{\beta} \) is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach.

      +

      where we have defined the matrix \( \boldsymbol{A} =\boldsymbol{X}/\boldsymbol{\Sigma} \) with matrix elements \( a_{ij} = x_{ij}/\sigma_i \) and the vector \( \boldsymbol{b} \) with elements \( b_i = y_i/\sigma_i \).

      -

      Let us now return to our nuclear binding energies and simply code the above equations.

      -

      diff --git a/doc/pub/week34/html/._week34-bs054.html b/doc/pub/week34/html/._week34-bs054.html index 9e1a65afe..18b48cda2 100644 --- a/doc/pub/week34/html/._week34-bs054.html +++ b/doc/pub/week34/html/._week34-bs054.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    9269. Installing R, C++, cython or Julia
    9270. Installing R, C++, cython, Numba etc
    9271. Numpy examples and Important Matrix and vector handling packages
    9272. -
    9273. Basic Matrix Features
    9274. -
    9275.    Some famous Matrices
    9276. -
    9277.    More Basic Matrix Features
    9278. -
    9279. Numpy and arrays
    9280. -
    9281. Matrices in Python
    9282. -
    9283. Meet the Pandas
    9284. -
    9285. Friday August 27
    9286. -
    9287.    Simple linear regression model using scikit-learn
    9288. -
    9289.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9290. -
    9291.    Organizing our data
    9292. -
    9293.    Seeing the wood for the trees
    9294. -
    9295.    And what about using neural networks?
    9296. -
    9297. A first summary
    9298. -
    9299. Why Linear Regression (aka Ordinary Least Squares and family)
    9300. -
    9301. Regression analysis, overarching aims
    9302. -
    9303. Regression analysis, overarching aims II
    9304. -
    9305. Examples
    9306. -
    9307. General linear models
    9308. -
    9309. Rewriting the fitting procedure as a linear algebra problem
    9310. -
    9311. Rewriting the fitting procedure as a linear algebra problem, more details
    9312. -
    9313. Generalizing the fitting procedure as a linear algebra problem
    9314. -
    9315. Generalizing the fitting procedure as a linear algebra problem
    9316. -
    9317. Optimizing our parameters
    9318. -
    9319. Our model for the nuclear binding energies
    9320. -
    9321. Optimizing our parameters, more details
    9322. -
    9323. Interpretations and optimizing our parameters
    9324. -
    9325. Interpretations and optimizing our parameters
    9326. -
    9327. Some useful matrix and vector expressions
    9328. -
    9329. Interpretations and optimizing our parameters
    9330. -
    9331. Own code for Ordinary Least Squares
    9332. -
    9333. Adding error analysis and training set up
    9334. -
    9335. The \( \chi^2 \) function
    9336. -
    9337. The \( \chi^2 \) function
    9338. -
    9339. The \( \chi^2 \) function
    9340. -
    9341. The \( \chi^2 \) function
    9342. -
    9343. The \( \chi^2 \) function
    9344. -
    9345. The \( \chi^2 \) function
    9346. -
    9347. Fitting an Equation of State for Dense Nuclear Matter
    9348. -
    9349. The code
    9350. -
    9351. Splitting our Data in Training and Test data
    9352. -
    9353. Exercises
    9354. -
    9355. Exercise 1: Setting up various Python environments
    9356. -
    9357. Exercise 2: making your own data and exploring scikit-learn
    9358. -
    9359. Exercise 3: Normalizing our data
    9360. +
    9361. Numpy and arrays
    9362. +
    9363. Matrices in Python
    9364. +
    9365. Meet the Pandas
    9366. +
    9367.    Simple linear regression model using scikit-learn
    9368. +
    9369.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9370. +
    9371.    Organizing our data
    9372. +
    9373.    And what about using neural networks?
    9374. +
    9375. A first summary
    9376. +
    9377. Why Linear Regression (aka Ordinary Least Squares and family)
    9378. +
    9379. Regression analysis, overarching aims
    9380. +
    9381. Regression analysis, overarching aims II
    9382. +
    9383. Examples
    9384. +
    9385. General linear models
    9386. +
    9387. Rewriting the fitting procedure as a linear algebra problem
    9388. +
    9389. Rewriting the fitting procedure as a linear algebra problem, more details
    9390. +
    9391. Generalizing the fitting procedure as a linear algebra problem
    9392. +
    9393. Generalizing the fitting procedure as a linear algebra problem
    9394. +
    9395. Optimizing our parameters
    9396. +
    9397. Our model for the nuclear binding energies
    9398. +
    9399. Optimizing our parameters, more details
    9400. +
    9401. Interpretations and optimizing our parameters
    9402. +
    9403. Interpretations and optimizing our parameters
    9404. +
    9405. Some useful matrix and vector expressions
    9406. +
    9407. Interpretations and optimizing our parameters
    9408. +
    9409. Own code for Ordinary Least Squares
    9410. +
    9411. Adding error analysis and training set up
    9412. +
    9413. The \( \chi^2 \) function
    9414. +
    9415. The \( \chi^2 \) function
    9416. +
    9417. The \( \chi^2 \) function
    9418. +
    9419. The \( \chi^2 \) function
    9420. +
    9421. The \( \chi^2 \) function
    9422. +
    9423. The \( \chi^2 \) function
    9424. +
    9425. Fitting an Equation of State for Dense Nuclear Matter
    9426. +
    9427. The code
    9428. +
    9429. Splitting our Data in Training and Test data
    9430. +
    9431. Exercises
    9432. +
    9433. Exercise 1: Setting up various Python environments
    9434. +
    9435. Exercise 2: making your own data and exploring scikit-learn
    9436. +
    9437. Exercise 3: Split data in test and training data
    9438. @@ -351,95 +335,26 @@ MathJax.Hub.Config({

       

       

       

      -

      Own code for Ordinary Least Squares

      +

      The \( \chi^2 \) function

      +
      +
      + -

      It is rather straightforward to implement the matrix inversion and obtain the parameters \( \boldsymbol{\beta} \). After having defined the matrix \( \boldsymbol{X} \) we simply need to -write -

      +

      We can rewrite

      +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right), +$$ - -
      -
      -
      -
      -
      -
      # matrix inversion to find beta
      -beta = np.linalg.inv(X.T.dot(X)).dot(X.T).dot(Energies)
      -# and then make the prediction
      -ytilde = X @ beta
      -
      +

      as

      +$$ +\boldsymbol{A}^T\boldsymbol{b} = \boldsymbol{A}^T\boldsymbol{A}\boldsymbol{\beta}, +$$ + +

      and if the matrix \( \boldsymbol{A}^T\boldsymbol{A} \) is invertible we have the solution

      +$$ +\boldsymbol{\beta} =\left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}\boldsymbol{A}^T\boldsymbol{b}. +$$
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Alternatively, you can use the least squares functionality in Numpy as

      - - -
      -
      -
      -
      -
      -
      fit = np.linalg.lstsq(X, Energies, rcond =None)[0]
      -ytildenp = np.dot(fit,X.T)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      And finally we plot our fit with and compare with data

      - - -
      -
      -
      -
      -
      -
      Masses['Eapprox']  = ytilde
      -# Generate a plot comparing the experimental with the fitted values values.
      -fig, ax = plt.subplots()
      -ax.set_xlabel(r'$A = N + Z$')
      -ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$')
      -ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,
      -            label='Ame2016')
      -ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',
      -            label='Fit')
      -ax.legend()
      -save_fig("Masses2016OLS")
      -plt.show()
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      @@ -465,10 +380,6 @@ plt.show()
    9439. 60
    9440. 61
    9441. 62
    9442. -
    9443. 63
    9444. -
    9445. 64
    9446. -
    9447. ...
    9448. -
    9449. 66
    9450. »
    9451. diff --git a/doc/pub/week34/html/._week34-bs055.html b/doc/pub/week34/html/._week34-bs055.html index b20679e85..723ad5e02 100644 --- a/doc/pub/week34/html/._week34-bs055.html +++ b/doc/pub/week34/html/._week34-bs055.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    9452. Installing R, C++, cython or Julia
    9453. Installing R, C++, cython, Numba etc
    9454. Numpy examples and Important Matrix and vector handling packages
    9455. -
    9456. Basic Matrix Features
    9457. -
    9458.    Some famous Matrices
    9459. -
    9460.    More Basic Matrix Features
    9461. -
    9462. Numpy and arrays
    9463. -
    9464. Matrices in Python
    9465. -
    9466. Meet the Pandas
    9467. -
    9468. Friday August 27
    9469. -
    9470.    Simple linear regression model using scikit-learn
    9471. -
    9472.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9473. -
    9474.    Organizing our data
    9475. -
    9476.    Seeing the wood for the trees
    9477. -
    9478.    And what about using neural networks?
    9479. -
    9480. A first summary
    9481. -
    9482. Why Linear Regression (aka Ordinary Least Squares and family)
    9483. -
    9484. Regression analysis, overarching aims
    9485. -
    9486. Regression analysis, overarching aims II
    9487. -
    9488. Examples
    9489. -
    9490. General linear models
    9491. -
    9492. Rewriting the fitting procedure as a linear algebra problem
    9493. -
    9494. Rewriting the fitting procedure as a linear algebra problem, more details
    9495. -
    9496. Generalizing the fitting procedure as a linear algebra problem
    9497. -
    9498. Generalizing the fitting procedure as a linear algebra problem
    9499. -
    9500. Optimizing our parameters
    9501. -
    9502. Our model for the nuclear binding energies
    9503. -
    9504. Optimizing our parameters, more details
    9505. -
    9506. Interpretations and optimizing our parameters
    9507. -
    9508. Interpretations and optimizing our parameters
    9509. -
    9510. Some useful matrix and vector expressions
    9511. -
    9512. Interpretations and optimizing our parameters
    9513. -
    9514. Own code for Ordinary Least Squares
    9515. -
    9516. Adding error analysis and training set up
    9517. -
    9518. The \( \chi^2 \) function
    9519. -
    9520. The \( \chi^2 \) function
    9521. -
    9522. The \( \chi^2 \) function
    9523. -
    9524. The \( \chi^2 \) function
    9525. -
    9526. The \( \chi^2 \) function
    9527. -
    9528. The \( \chi^2 \) function
    9529. -
    9530. Fitting an Equation of State for Dense Nuclear Matter
    9531. -
    9532. The code
    9533. -
    9534. Splitting our Data in Training and Test data
    9535. -
    9536. Exercises
    9537. -
    9538. Exercise 1: Setting up various Python environments
    9539. -
    9540. Exercise 2: making your own data and exploring scikit-learn
    9541. -
    9542. Exercise 3: Normalizing our data
    9543. +
    9544. Numpy and arrays
    9545. +
    9546. Matrices in Python
    9547. +
    9548. Meet the Pandas
    9549. +
    9550.    Simple linear regression model using scikit-learn
    9551. +
    9552.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9553. +
    9554.    Organizing our data
    9555. +
    9556.    And what about using neural networks?
    9557. +
    9558. A first summary
    9559. +
    9560. Why Linear Regression (aka Ordinary Least Squares and family)
    9561. +
    9562. Regression analysis, overarching aims
    9563. +
    9564. Regression analysis, overarching aims II
    9565. +
    9566. Examples
    9567. +
    9568. General linear models
    9569. +
    9570. Rewriting the fitting procedure as a linear algebra problem
    9571. +
    9572. Rewriting the fitting procedure as a linear algebra problem, more details
    9573. +
    9574. Generalizing the fitting procedure as a linear algebra problem
    9575. +
    9576. Generalizing the fitting procedure as a linear algebra problem
    9577. +
    9578. Optimizing our parameters
    9579. +
    9580. Our model for the nuclear binding energies
    9581. +
    9582. Optimizing our parameters, more details
    9583. +
    9584. Interpretations and optimizing our parameters
    9585. +
    9586. Interpretations and optimizing our parameters
    9587. +
    9588. Some useful matrix and vector expressions
    9589. +
    9590. Interpretations and optimizing our parameters
    9591. +
    9592. Own code for Ordinary Least Squares
    9593. +
    9594. Adding error analysis and training set up
    9595. +
    9596. The \( \chi^2 \) function
    9597. +
    9598. The \( \chi^2 \) function
    9599. +
    9600. The \( \chi^2 \) function
    9601. +
    9602. The \( \chi^2 \) function
    9603. +
    9604. The \( \chi^2 \) function
    9605. +
    9606. The \( \chi^2 \) function
    9607. +
    9608. Fitting an Equation of State for Dense Nuclear Matter
    9609. +
    9610. The code
    9611. +
    9612. Splitting our Data in Training and Test data
    9613. +
    9614. Exercises
    9615. +
    9616. Exercise 1: Setting up various Python environments
    9617. +
    9618. Exercise 2: making your own data and exploring scikit-learn
    9619. +
    9620. Exercise 3: Split data in test and training data
    9621. @@ -351,111 +335,31 @@ MathJax.Hub.Config({

       

       

       

      -

      Adding error analysis and training set up

      +

      The \( \chi^2 \) function

      +
      +
      + -

      We can easily test our fit by computing the \( R2 \) score that we discussed in connection with the functionality of Scikit-Learn in the introductory slides. -Since we are not using Scikit-Learn here we can define our own \( R2 \) function as -

      +

      If we then introduce the matrix

      +$$ +\boldsymbol{H} = \left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}, +$$ - -
      -
      -
      -
      -
      -
      def R2(y_data, y_model):
      -    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
      -
      +

      we have then the following expression for the parameters \( \beta_j \) (the matrix elements of \( \boldsymbol{H} \) are \( h_{ij} \))

      +$$ +\beta_j = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}\frac{y_i}{\sigma_i}\frac{x_{ik}}{\sigma_i} = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}b_ia_{ik} +$$ + +

      We state without proof the expression for the uncertainty in the parameters \( \beta_j \) as (we leave this as an exercise)

      +$$ +\sigma^2(\beta_j) = \sum_{i=0}^{n-1}\sigma_i^2\left( \frac{\partial \beta_j}{\partial y_i}\right)^2, +$$ + +

      resulting in

      +$$ +\sigma^2(\beta_j) = \left(\sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}a_{ik}\right)\left(\sum_{l=0}^{p-1}h_{jl}\sum_{m=0}^{n-1}a_{ml}\right) = h_{jj}! +$$
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      and we would be using it as

      - - -
      -
      -
      -
      -
      -
      print(R2(Energies,ytilde))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      We can easily add our MSE score as

      - - -
      -
      -
      -
      -
      -
      def MSE(y_data,y_model):
      -    n = np.size(y_model)
      -    return np.sum((y_data-y_model)**2)/n
      -
      -print(MSE(Energies,ytilde))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      and finally the relative error as

      - - -
      -
      -
      -
      -
      -
      def RelativeError(y_data,y_model):
      -    return abs((y_data-y_model)/y_data)
      -print(RelativeError(Energies, ytilde))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      @@ -480,11 +384,6 @@ Since we are not using Scikit-Learn here we can define our own \( R2 \) f
    9622. 60
    9623. 61
    9624. 62
    9625. -
    9626. 63
    9627. -
    9628. 64
    9629. -
    9630. 65
    9631. -
    9632. ...
    9633. -
    9634. 66
    9635. »
    9636. diff --git a/doc/pub/week34/html/._week34-bs056.html b/doc/pub/week34/html/._week34-bs056.html index 97af0e1c6..42efdd668 100644 --- a/doc/pub/week34/html/._week34-bs056.html +++ b/doc/pub/week34/html/._week34-bs056.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    9637. Installing R, C++, cython or Julia
    9638. Installing R, C++, cython, Numba etc
    9639. Numpy examples and Important Matrix and vector handling packages
    9640. -
    9641. Basic Matrix Features
    9642. -
    9643.    Some famous Matrices
    9644. -
    9645.    More Basic Matrix Features
    9646. -
    9647. Numpy and arrays
    9648. -
    9649. Matrices in Python
    9650. -
    9651. Meet the Pandas
    9652. -
    9653. Friday August 27
    9654. -
    9655.    Simple linear regression model using scikit-learn
    9656. -
    9657.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9658. -
    9659.    Organizing our data
    9660. -
    9661.    Seeing the wood for the trees
    9662. -
    9663.    And what about using neural networks?
    9664. -
    9665. A first summary
    9666. -
    9667. Why Linear Regression (aka Ordinary Least Squares and family)
    9668. -
    9669. Regression analysis, overarching aims
    9670. -
    9671. Regression analysis, overarching aims II
    9672. -
    9673. Examples
    9674. -
    9675. General linear models
    9676. -
    9677. Rewriting the fitting procedure as a linear algebra problem
    9678. -
    9679. Rewriting the fitting procedure as a linear algebra problem, more details
    9680. -
    9681. Generalizing the fitting procedure as a linear algebra problem
    9682. -
    9683. Generalizing the fitting procedure as a linear algebra problem
    9684. -
    9685. Optimizing our parameters
    9686. -
    9687. Our model for the nuclear binding energies
    9688. -
    9689. Optimizing our parameters, more details
    9690. -
    9691. Interpretations and optimizing our parameters
    9692. -
    9693. Interpretations and optimizing our parameters
    9694. -
    9695. Some useful matrix and vector expressions
    9696. -
    9697. Interpretations and optimizing our parameters
    9698. -
    9699. Own code for Ordinary Least Squares
    9700. -
    9701. Adding error analysis and training set up
    9702. -
    9703. The \( \chi^2 \) function
    9704. -
    9705. The \( \chi^2 \) function
    9706. -
    9707. The \( \chi^2 \) function
    9708. -
    9709. The \( \chi^2 \) function
    9710. -
    9711. The \( \chi^2 \) function
    9712. -
    9713. The \( \chi^2 \) function
    9714. -
    9715. Fitting an Equation of State for Dense Nuclear Matter
    9716. -
    9717. The code
    9718. -
    9719. Splitting our Data in Training and Test data
    9720. -
    9721. Exercises
    9722. -
    9723. Exercise 1: Setting up various Python environments
    9724. -
    9725. Exercise 2: making your own data and exploring scikit-learn
    9726. -
    9727. Exercise 3: Normalizing our data
    9728. +
    9729. Numpy and arrays
    9730. +
    9731. Matrices in Python
    9732. +
    9733. Meet the Pandas
    9734. +
    9735.    Simple linear regression model using scikit-learn
    9736. +
    9737.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9738. +
    9739.    Organizing our data
    9740. +
    9741.    And what about using neural networks?
    9742. +
    9743. A first summary
    9744. +
    9745. Why Linear Regression (aka Ordinary Least Squares and family)
    9746. +
    9747. Regression analysis, overarching aims
    9748. +
    9749. Regression analysis, overarching aims II
    9750. +
    9751. Examples
    9752. +
    9753. General linear models
    9754. +
    9755. Rewriting the fitting procedure as a linear algebra problem
    9756. +
    9757. Rewriting the fitting procedure as a linear algebra problem, more details
    9758. +
    9759. Generalizing the fitting procedure as a linear algebra problem
    9760. +
    9761. Generalizing the fitting procedure as a linear algebra problem
    9762. +
    9763. Optimizing our parameters
    9764. +
    9765. Our model for the nuclear binding energies
    9766. +
    9767. Optimizing our parameters, more details
    9768. +
    9769. Interpretations and optimizing our parameters
    9770. +
    9771. Interpretations and optimizing our parameters
    9772. +
    9773. Some useful matrix and vector expressions
    9774. +
    9775. Interpretations and optimizing our parameters
    9776. +
    9777. Own code for Ordinary Least Squares
    9778. +
    9779. Adding error analysis and training set up
    9780. +
    9781. The \( \chi^2 \) function
    9782. +
    9783. The \( \chi^2 \) function
    9784. +
    9785. The \( \chi^2 \) function
    9786. +
    9787. The \( \chi^2 \) function
    9788. +
    9789. The \( \chi^2 \) function
    9790. +
    9791. The \( \chi^2 \) function
    9792. +
    9793. Fitting an Equation of State for Dense Nuclear Matter
    9794. +
    9795. The code
    9796. +
    9797. Splitting our Data in Training and Test data
    9798. +
    9799. Exercises
    9800. +
    9801. Exercise 1: Setting up various Python environments
    9802. +
    9803. Exercise 2: making your own data and exploring scikit-learn
    9804. +
    9805. Exercise 3: Split data in test and training data
    9806. @@ -355,26 +339,20 @@ MathJax.Hub.Config({
      - -

      Normally, the response (dependent or outcome) variable \( y_i \) is the -outcome of a numerical experiment or another type of experiment and is -thus only an approximation to the true value. It is then always -accompanied by an error estimate, often limited to a statistical error -estimate given by the standard deviation discussed earlier. In the -discussion here we will treat \( y_i \) as our exact value for the -response variable. -

      - -

      Introducing the standard deviation \( \sigma_i \) for each measurement -\( y_i \), we define now the \( \chi^2 \) function (omitting the \( 1/n \) term) -as -

      - +

      The first step here is to approximate the function \( y \) with a first-order polynomial, that is we write

      $$ -\chi^2(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\frac{\left(y_i-\tilde{y}_i\right)^2}{\sigma_i^2}=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\frac{1}{\boldsymbol{\Sigma^2}}\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +y=y(x) \rightarrow y(x_i) \approx \beta_0+\beta_1 x_i. $$ -

      where the matrix \( \boldsymbol{\Sigma} \) is a diagonal matrix with \( \sigma_i \) as matrix elements.

      +

      By computing the derivatives of \( \chi^2 \) with respect to \( \beta_0 \) and \( \beta_1 \) show that these are given by

      +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_0} = -2\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0, +$$ + +

      and

      +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_1} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_i\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0. +$$
      @@ -399,10 +377,6 @@ $$
    9807. 60
    9808. 61
    9809. 62
    9810. -
    9811. 63
    9812. -
    9813. 64
    9814. -
    9815. 65
    9816. -
    9817. 66
    9818. »
    9819. diff --git a/doc/pub/week34/html/._week34-bs057.html b/doc/pub/week34/html/._week34-bs057.html index 53cf16b1b..4b7b322f7 100644 --- a/doc/pub/week34/html/._week34-bs057.html +++ b/doc/pub/week34/html/._week34-bs057.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    9820. Installing R, C++, cython or Julia
    9821. Installing R, C++, cython, Numba etc
    9822. Numpy examples and Important Matrix and vector handling packages
    9823. -
    9824. Basic Matrix Features
    9825. -
    9826.    Some famous Matrices
    9827. -
    9828.    More Basic Matrix Features
    9829. -
    9830. Numpy and arrays
    9831. -
    9832. Matrices in Python
    9833. -
    9834. Meet the Pandas
    9835. -
    9836. Friday August 27
    9837. -
    9838.    Simple linear regression model using scikit-learn
    9839. -
    9840.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9841. -
    9842.    Organizing our data
    9843. -
    9844.    Seeing the wood for the trees
    9845. -
    9846.    And what about using neural networks?
    9847. -
    9848. A first summary
    9849. -
    9850. Why Linear Regression (aka Ordinary Least Squares and family)
    9851. -
    9852. Regression analysis, overarching aims
    9853. -
    9854. Regression analysis, overarching aims II
    9855. -
    9856. Examples
    9857. -
    9858. General linear models
    9859. -
    9860. Rewriting the fitting procedure as a linear algebra problem
    9861. -
    9862. Rewriting the fitting procedure as a linear algebra problem, more details
    9863. -
    9864. Generalizing the fitting procedure as a linear algebra problem
    9865. -
    9866. Generalizing the fitting procedure as a linear algebra problem
    9867. -
    9868. Optimizing our parameters
    9869. -
    9870. Our model for the nuclear binding energies
    9871. -
    9872. Optimizing our parameters, more details
    9873. -
    9874. Interpretations and optimizing our parameters
    9875. -
    9876. Interpretations and optimizing our parameters
    9877. -
    9878. Some useful matrix and vector expressions
    9879. -
    9880. Interpretations and optimizing our parameters
    9881. -
    9882. Own code for Ordinary Least Squares
    9883. -
    9884. Adding error analysis and training set up
    9885. -
    9886. The \( \chi^2 \) function
    9887. -
    9888. The \( \chi^2 \) function
    9889. -
    9890. The \( \chi^2 \) function
    9891. -
    9892. The \( \chi^2 \) function
    9893. -
    9894. The \( \chi^2 \) function
    9895. -
    9896. The \( \chi^2 \) function
    9897. -
    9898. Fitting an Equation of State for Dense Nuclear Matter
    9899. -
    9900. The code
    9901. -
    9902. Splitting our Data in Training and Test data
    9903. -
    9904. Exercises
    9905. -
    9906. Exercise 1: Setting up various Python environments
    9907. -
    9908. Exercise 2: making your own data and exploring scikit-learn
    9909. -
    9910. Exercise 3: Normalizing our data
    9911. +
    9912. Numpy and arrays
    9913. +
    9914. Matrices in Python
    9915. +
    9916. Meet the Pandas
    9917. +
    9918.    Simple linear regression model using scikit-learn
    9919. +
    9920.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    9921. +
    9922.    Organizing our data
    9923. +
    9924.    And what about using neural networks?
    9925. +
    9926. A first summary
    9927. +
    9928. Why Linear Regression (aka Ordinary Least Squares and family)
    9929. +
    9930. Regression analysis, overarching aims
    9931. +
    9932. Regression analysis, overarching aims II
    9933. +
    9934. Examples
    9935. +
    9936. General linear models
    9937. +
    9938. Rewriting the fitting procedure as a linear algebra problem
    9939. +
    9940. Rewriting the fitting procedure as a linear algebra problem, more details
    9941. +
    9942. Generalizing the fitting procedure as a linear algebra problem
    9943. +
    9944. Generalizing the fitting procedure as a linear algebra problem
    9945. +
    9946. Optimizing our parameters
    9947. +
    9948. Our model for the nuclear binding energies
    9949. +
    9950. Optimizing our parameters, more details
    9951. +
    9952. Interpretations and optimizing our parameters
    9953. +
    9954. Interpretations and optimizing our parameters
    9955. +
    9956. Some useful matrix and vector expressions
    9957. +
    9958. Interpretations and optimizing our parameters
    9959. +
    9960. Own code for Ordinary Least Squares
    9961. +
    9962. Adding error analysis and training set up
    9963. +
    9964. The \( \chi^2 \) function
    9965. +
    9966. The \( \chi^2 \) function
    9967. +
    9968. The \( \chi^2 \) function
    9969. +
    9970. The \( \chi^2 \) function
    9971. +
    9972. The \( \chi^2 \) function
    9973. +
    9974. The \( \chi^2 \) function
    9975. +
    9976. Fitting an Equation of State for Dense Nuclear Matter
    9977. +
    9978. The code
    9979. +
    9980. Splitting our Data in Training and Test data
    9981. +
    9982. Exercises
    9983. +
    9984. Exercise 1: Setting up various Python environments
    9985. +
    9986. Exercise 2: making your own data and exploring scikit-learn
    9987. +
    9988. Exercise 3: Split data in test and training data
    9989. @@ -356,22 +340,49 @@ MathJax.Hub.Config({
      -

      In order to find the parameters \( \beta_i \) we will then minimize the spread of \( \chi^2(\boldsymbol{\beta}) \) by requiring

      +

      For a linear fit (a first-order polynomial) we don't need to invert a matrix!! +Defining +

      $$ -\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)^2\right]=0, +\gamma = \sum_{i=0}^{n-1}\frac{1}{\sigma_i^2}, $$ -

      which results in

      + $$ -\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}\frac{x_{ij}}{\sigma_i}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)\right]=0, +\gamma_x = \sum_{i=0}^{n-1}\frac{x_{i}}{\sigma_i^2}, $$ -

      or in a matrix-vector form as

      + $$ -\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right). +\gamma_y = \sum_{i=0}^{n-1}\left(\frac{y_i}{\sigma_i^2}\right), $$ -

      where we have defined the matrix \( \boldsymbol{A} =\boldsymbol{X}/\boldsymbol{\Sigma} \) with matrix elements \( a_{ij} = x_{ij}/\sigma_i \) and the vector \( \boldsymbol{b} \) with elements \( b_i = y_i/\sigma_i \).

      + +$$ +\gamma_{xx} = \sum_{i=0}^{n-1}\frac{x_ix_{i}}{\sigma_i^2}, +$$ + + +$$ +\gamma_{xy} = \sum_{i=0}^{n-1}\frac{y_ix_{i}}{\sigma_i^2}, +$$ + +

      we obtain

      + +$$ +\beta_0 = \frac{\gamma_{xx}\gamma_y-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}, +$$ + + +$$ +\beta_1 = \frac{\gamma_{xy}\gamma-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}. +$$ + +

      This approach (different linear and non-linear regression) suffers +often from both being underdetermined and overdetermined in the +unknown coefficients \( \beta_i \). A better approach is to use the +Singular Value Decomposition (SVD) method discussed next week. +

      @@ -395,10 +406,6 @@ $$
    9990. 60
    9991. 61
    9992. 62
    9993. -
    9994. 63
    9995. -
    9996. 64
    9997. -
    9998. 65
    9999. -
    10000. 66
    10001. »
    10002. diff --git a/doc/pub/week34/html/._week34-bs058.html b/doc/pub/week34/html/._week34-bs058.html index 447ed05cf..b07af483e 100644 --- a/doc/pub/week34/html/._week34-bs058.html +++ b/doc/pub/week34/html/._week34-bs058.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    10003. Installing R, C++, cython or Julia
    10004. Installing R, C++, cython, Numba etc
    10005. Numpy examples and Important Matrix and vector handling packages
    10006. -
    10007. Basic Matrix Features
    10008. -
    10009.    Some famous Matrices
    10010. -
    10011.    More Basic Matrix Features
    10012. -
    10013. Numpy and arrays
    10014. -
    10015. Matrices in Python
    10016. -
    10017. Meet the Pandas
    10018. -
    10019. Friday August 27
    10020. -
    10021.    Simple linear regression model using scikit-learn
    10022. -
    10023.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10024. -
    10025.    Organizing our data
    10026. -
    10027.    Seeing the wood for the trees
    10028. -
    10029.    And what about using neural networks?
    10030. -
    10031. A first summary
    10032. -
    10033. Why Linear Regression (aka Ordinary Least Squares and family)
    10034. -
    10035. Regression analysis, overarching aims
    10036. -
    10037. Regression analysis, overarching aims II
    10038. -
    10039. Examples
    10040. -
    10041. General linear models
    10042. -
    10043. Rewriting the fitting procedure as a linear algebra problem
    10044. -
    10045. Rewriting the fitting procedure as a linear algebra problem, more details
    10046. -
    10047. Generalizing the fitting procedure as a linear algebra problem
    10048. -
    10049. Generalizing the fitting procedure as a linear algebra problem
    10050. -
    10051. Optimizing our parameters
    10052. -
    10053. Our model for the nuclear binding energies
    10054. -
    10055. Optimizing our parameters, more details
    10056. -
    10057. Interpretations and optimizing our parameters
    10058. -
    10059. Interpretations and optimizing our parameters
    10060. -
    10061. Some useful matrix and vector expressions
    10062. -
    10063. Interpretations and optimizing our parameters
    10064. -
    10065. Own code for Ordinary Least Squares
    10066. -
    10067. Adding error analysis and training set up
    10068. -
    10069. The \( \chi^2 \) function
    10070. -
    10071. The \( \chi^2 \) function
    10072. -
    10073. The \( \chi^2 \) function
    10074. -
    10075. The \( \chi^2 \) function
    10076. -
    10077. The \( \chi^2 \) function
    10078. -
    10079. The \( \chi^2 \) function
    10080. -
    10081. Fitting an Equation of State for Dense Nuclear Matter
    10082. -
    10083. The code
    10084. -
    10085. Splitting our Data in Training and Test data
    10086. -
    10087. Exercises
    10088. -
    10089. Exercise 1: Setting up various Python environments
    10090. -
    10091. Exercise 2: making your own data and exploring scikit-learn
    10092. -
    10093. Exercise 3: Normalizing our data
    10094. +
    10095. Numpy and arrays
    10096. +
    10097. Matrices in Python
    10098. +
    10099. Meet the Pandas
    10100. +
    10101.    Simple linear regression model using scikit-learn
    10102. +
    10103.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10104. +
    10105.    Organizing our data
    10106. +
    10107.    And what about using neural networks?
    10108. +
    10109. A first summary
    10110. +
    10111. Why Linear Regression (aka Ordinary Least Squares and family)
    10112. +
    10113. Regression analysis, overarching aims
    10114. +
    10115. Regression analysis, overarching aims II
    10116. +
    10117. Examples
    10118. +
    10119. General linear models
    10120. +
    10121. Rewriting the fitting procedure as a linear algebra problem
    10122. +
    10123. Rewriting the fitting procedure as a linear algebra problem, more details
    10124. +
    10125. Generalizing the fitting procedure as a linear algebra problem
    10126. +
    10127. Generalizing the fitting procedure as a linear algebra problem
    10128. +
    10129. Optimizing our parameters
    10130. +
    10131. Our model for the nuclear binding energies
    10132. +
    10133. Optimizing our parameters, more details
    10134. +
    10135. Interpretations and optimizing our parameters
    10136. +
    10137. Interpretations and optimizing our parameters
    10138. +
    10139. Some useful matrix and vector expressions
    10140. +
    10141. Interpretations and optimizing our parameters
    10142. +
    10143. Own code for Ordinary Least Squares
    10144. +
    10145. Adding error analysis and training set up
    10146. +
    10147. The \( \chi^2 \) function
    10148. +
    10149. The \( \chi^2 \) function
    10150. +
    10151. The \( \chi^2 \) function
    10152. +
    10153. The \( \chi^2 \) function
    10154. +
    10155. The \( \chi^2 \) function
    10156. +
    10157. The \( \chi^2 \) function
    10158. +
    10159. Fitting an Equation of State for Dense Nuclear Matter
    10160. +
    10161. The code
    10162. +
    10163. Splitting our Data in Training and Test data
    10164. +
    10165. Exercises
    10166. +
    10167. Exercise 1: Setting up various Python environments
    10168. +
    10169. Exercise 2: making your own data and exploring scikit-learn
    10170. +
    10171. Exercise 3: Split data in test and training data
    10172. @@ -351,28 +335,27 @@ MathJax.Hub.Config({

       

       

       

      -

      The \( \chi^2 \) function

      -
      -
      - +

      Fitting an Equation of State for Dense Nuclear Matter

      -

      We can rewrite

      -$$ -\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right), -$$ +

      Before we continue, let us introduce yet another example. We are going to fit the +nuclear equation of state using results from many-body calculations. +The equation of state we have made available here, as function of +density, has been derived using modern nucleon-nucleon potentials with +the addition of three-body +forces. This +time the file is presented as a standard csv file. +

      -

      as

      -$$ -\boldsymbol{A}^T\boldsymbol{b} = \boldsymbol{A}^T\boldsymbol{A}\boldsymbol{\beta}, -$$ - -

      and if the matrix \( \boldsymbol{A}^T\boldsymbol{A} \) is invertible we have the solution

      -$$ -\boldsymbol{\beta} =\left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}\boldsymbol{A}^T\boldsymbol{b}. -$$ -
      -
      +

      The beginning of the Python code here is similar to what you have seen +before, with the same initializations and declarations. We use also +pandas again, rather extensively in order to organize our data. +

      +

      The difference now is that we use Scikit-Learn's regression tools +instead of our own matrix inversion implementation. Furthermore, we +sneak in Ridge regression (to be discussed below) which includes a +hyperparameter \( \lambda \), also to be explained below. +

      @@ -392,10 +375,6 @@ $$

    10173. 60
    10174. 61
    10175. 62
    10176. -
    10177. 63
    10178. -
    10179. 64
    10180. -
    10181. 65
    10182. -
    10183. 66
    10184. »
    10185. diff --git a/doc/pub/week34/html/._week34-bs059.html b/doc/pub/week34/html/._week34-bs059.html index 0eabb58c4..0dd2d071f 100644 --- a/doc/pub/week34/html/._week34-bs059.html +++ b/doc/pub/week34/html/._week34-bs059.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    10186. Installing R, C++, cython or Julia
    10187. Installing R, C++, cython, Numba etc
    10188. Numpy examples and Important Matrix and vector handling packages
    10189. -
    10190. Basic Matrix Features
    10191. -
    10192.    Some famous Matrices
    10193. -
    10194.    More Basic Matrix Features
    10195. -
    10196. Numpy and arrays
    10197. -
    10198. Matrices in Python
    10199. -
    10200. Meet the Pandas
    10201. -
    10202. Friday August 27
    10203. -
    10204.    Simple linear regression model using scikit-learn
    10205. -
    10206.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10207. -
    10208.    Organizing our data
    10209. -
    10210.    Seeing the wood for the trees
    10211. -
    10212.    And what about using neural networks?
    10213. -
    10214. A first summary
    10215. -
    10216. Why Linear Regression (aka Ordinary Least Squares and family)
    10217. -
    10218. Regression analysis, overarching aims
    10219. -
    10220. Regression analysis, overarching aims II
    10221. -
    10222. Examples
    10223. -
    10224. General linear models
    10225. -
    10226. Rewriting the fitting procedure as a linear algebra problem
    10227. -
    10228. Rewriting the fitting procedure as a linear algebra problem, more details
    10229. -
    10230. Generalizing the fitting procedure as a linear algebra problem
    10231. -
    10232. Generalizing the fitting procedure as a linear algebra problem
    10233. -
    10234. Optimizing our parameters
    10235. -
    10236. Our model for the nuclear binding energies
    10237. -
    10238. Optimizing our parameters, more details
    10239. -
    10240. Interpretations and optimizing our parameters
    10241. -
    10242. Interpretations and optimizing our parameters
    10243. -
    10244. Some useful matrix and vector expressions
    10245. -
    10246. Interpretations and optimizing our parameters
    10247. -
    10248. Own code for Ordinary Least Squares
    10249. -
    10250. Adding error analysis and training set up
    10251. -
    10252. The \( \chi^2 \) function
    10253. -
    10254. The \( \chi^2 \) function
    10255. -
    10256. The \( \chi^2 \) function
    10257. -
    10258. The \( \chi^2 \) function
    10259. -
    10260. The \( \chi^2 \) function
    10261. -
    10262. The \( \chi^2 \) function
    10263. -
    10264. Fitting an Equation of State for Dense Nuclear Matter
    10265. -
    10266. The code
    10267. -
    10268. Splitting our Data in Training and Test data
    10269. -
    10270. Exercises
    10271. -
    10272. Exercise 1: Setting up various Python environments
    10273. -
    10274. Exercise 2: making your own data and exploring scikit-learn
    10275. -
    10276. Exercise 3: Normalizing our data
    10277. +
    10278. Numpy and arrays
    10279. +
    10280. Matrices in Python
    10281. +
    10282. Meet the Pandas
    10283. +
    10284.    Simple linear regression model using scikit-learn
    10285. +
    10286.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10287. +
    10288.    Organizing our data
    10289. +
    10290.    And what about using neural networks?
    10291. +
    10292. A first summary
    10293. +
    10294. Why Linear Regression (aka Ordinary Least Squares and family)
    10295. +
    10296. Regression analysis, overarching aims
    10297. +
    10298. Regression analysis, overarching aims II
    10299. +
    10300. Examples
    10301. +
    10302. General linear models
    10303. +
    10304. Rewriting the fitting procedure as a linear algebra problem
    10305. +
    10306. Rewriting the fitting procedure as a linear algebra problem, more details
    10307. +
    10308. Generalizing the fitting procedure as a linear algebra problem
    10309. +
    10310. Generalizing the fitting procedure as a linear algebra problem
    10311. +
    10312. Optimizing our parameters
    10313. +
    10314. Our model for the nuclear binding energies
    10315. +
    10316. Optimizing our parameters, more details
    10317. +
    10318. Interpretations and optimizing our parameters
    10319. +
    10320. Interpretations and optimizing our parameters
    10321. +
    10322. Some useful matrix and vector expressions
    10323. +
    10324. Interpretations and optimizing our parameters
    10325. +
    10326. Own code for Ordinary Least Squares
    10327. +
    10328. Adding error analysis and training set up
    10329. +
    10330. The \( \chi^2 \) function
    10331. +
    10332. The \( \chi^2 \) function
    10333. +
    10334. The \( \chi^2 \) function
    10335. +
    10336. The \( \chi^2 \) function
    10337. +
    10338. The \( \chi^2 \) function
    10339. +
    10340. The \( \chi^2 \) function
    10341. +
    10342. Fitting an Equation of State for Dense Nuclear Matter
    10343. +
    10344. The code
    10345. +
    10346. Splitting our Data in Training and Test data
    10347. +
    10348. Exercises
    10349. +
    10350. Exercise 1: Setting up various Python environments
    10351. +
    10352. Exercise 2: making your own data and exploring scikit-learn
    10353. +
    10354. Exercise 3: Split data in test and training data
    10355. @@ -351,33 +335,123 @@ MathJax.Hub.Config({

       

       

       

      -

      The \( \chi^2 \) function

      -
      -
      - +

      The code

      -

      If we then introduce the matrix

      -$$ -\boldsymbol{H} = \left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}, -$$ -

      we have then the following expression for the parameters \( \beta_j \) (the matrix elements of \( \boldsymbol{H} \) are \( h_{ij} \))

      -$$ -\beta_j = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}\frac{y_i}{\sigma_i}\frac{x_{ik}}{\sigma_i} = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}b_ia_{ik} -$$ + +
      +
      +
      +
      +
      +
      # Common imports
      +import os
      +import numpy as np
      +import pandas as pd
      +import matplotlib.pyplot as plt
      +import matplotlib.pyplot as plt
      +import sklearn.linear_model as skl
      +from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
       
      -

      We state without proof the expression for the uncertainty in the parameters \( \beta_j \) as (we leave this as an exercise)

      -$$ -\sigma^2(\beta_j) = \sum_{i=0}^{n-1}\sigma_i^2\left( \frac{\partial \beta_j}{\partial y_i}\right)^2, -$$ +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" -

      resulting in

      -$$ -\sigma^2(\beta_j) = \left(\sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}a_{ik}\right)\left(\sum_{l=0}^{p-1}h_{jl}\sum_{m=0}^{n-1}a_{ml}\right) = h_{jj}! -$$ +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("EoS.csv"),'r') + +# Read the EoS data as csv file and organize the data into two arrays with density and energies +EoS = pd.read_csv(infile, names=('Density', 'Energy')) +EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce') +EoS = EoS.dropna() +Energies = EoS['Energy'] +Density = EoS['Density'] +# The design matrix now as function of various polytrops +X = np.zeros((len(Density),4)) +X[:,3] = Density**(4.0/3.0) +X[:,2] = Density +X[:,1] = Density**(2.0/3.0) +X[:,0] = 1 + +# We use now Scikit-Learn's linear regressor and ridge regressor +# OLS part +clf = skl.LinearRegression().fit(X, Energies) +ytilde = clf.predict(X) +EoS['Eols'] = ytilde +# The mean squared error +print("Mean squared error: %.2f" % mean_squared_error(Energies, ytilde)) +# Explained variance score: 1 is perfect prediction +print('Variance score: %.2f' % r2_score(Energies, ytilde)) +# Mean absolute error +print('Mean absolute error: %.2f' % mean_absolute_error(Energies, ytilde)) +print(clf.coef_, clf.intercept_) + +# The Ridge regression with a hyperparameter lambda = 0.1 +_lambda = 0.1 +clf_ridge = skl.Ridge(alpha=_lambda).fit(X, Energies) +yridge = clf_ridge.predict(X) +EoS['Eridge'] = yridge +# The mean squared error +print("Mean squared error: %.2f" % mean_squared_error(Energies, yridge)) +# Explained variance score: 1 is perfect prediction +print('Variance score: %.2f' % r2_score(Energies, yridge)) +# Mean absolute error +print('Mean absolute error: %.2f' % mean_absolute_error(Energies, yridge)) +print(clf_ridge.coef_, clf_ridge.intercept_) + +fig, ax = plt.subplots() +ax.set_xlabel(r'$\rho[\mathrm{fm}^{-3}]$') +ax.set_ylabel(r'Energy per particle') +ax.plot(EoS['Density'], EoS['Energy'], alpha=0.7, lw=2, + label='Theoretical data') +ax.plot(EoS['Density'], EoS['Eols'], alpha=0.7, lw=2, c='m', + label='OLS') +ax.plot(EoS['Density'], EoS['Eridge'], alpha=0.7, lw=2, c='g', + label='Ridge $\lambda = 0.1$') +ax.legend() +save_fig("EoSfitting") +plt.show() +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +

      The above simple polynomial in density \( \rho \) gives an excellent fit +to the data. +

      + +

      We note also that there is a small deviation between the +standard OLS and the Ridge regression at higher densities. We discuss this in more detail +below. +

      @@ -396,10 +470,6 @@ $$

    10356. 60
    10357. 61
    10358. 62
    10359. -
    10360. 63
    10361. -
    10362. 64
    10363. -
    10364. 65
    10365. -
    10366. 66
    10367. »
    10368. diff --git a/doc/pub/week34/html/._week34-bs060.html b/doc/pub/week34/html/._week34-bs060.html index aba97711b..60e7d6219 100644 --- a/doc/pub/week34/html/._week34-bs060.html +++ b/doc/pub/week34/html/._week34-bs060.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    10369. Installing R, C++, cython or Julia
    10370. Installing R, C++, cython, Numba etc
    10371. Numpy examples and Important Matrix and vector handling packages
    10372. -
    10373. Basic Matrix Features
    10374. -
    10375.    Some famous Matrices
    10376. -
    10377.    More Basic Matrix Features
    10378. -
    10379. Numpy and arrays
    10380. -
    10381. Matrices in Python
    10382. -
    10383. Meet the Pandas
    10384. -
    10385. Friday August 27
    10386. -
    10387.    Simple linear regression model using scikit-learn
    10388. -
    10389.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10390. -
    10391.    Organizing our data
    10392. -
    10393.    Seeing the wood for the trees
    10394. -
    10395.    And what about using neural networks?
    10396. -
    10397. A first summary
    10398. -
    10399. Why Linear Regression (aka Ordinary Least Squares and family)
    10400. -
    10401. Regression analysis, overarching aims
    10402. -
    10403. Regression analysis, overarching aims II
    10404. -
    10405. Examples
    10406. -
    10407. General linear models
    10408. -
    10409. Rewriting the fitting procedure as a linear algebra problem
    10410. -
    10411. Rewriting the fitting procedure as a linear algebra problem, more details
    10412. -
    10413. Generalizing the fitting procedure as a linear algebra problem
    10414. -
    10415. Generalizing the fitting procedure as a linear algebra problem
    10416. -
    10417. Optimizing our parameters
    10418. -
    10419. Our model for the nuclear binding energies
    10420. -
    10421. Optimizing our parameters, more details
    10422. -
    10423. Interpretations and optimizing our parameters
    10424. -
    10425. Interpretations and optimizing our parameters
    10426. -
    10427. Some useful matrix and vector expressions
    10428. -
    10429. Interpretations and optimizing our parameters
    10430. -
    10431. Own code for Ordinary Least Squares
    10432. -
    10433. Adding error analysis and training set up
    10434. -
    10435. The \( \chi^2 \) function
    10436. -
    10437. The \( \chi^2 \) function
    10438. -
    10439. The \( \chi^2 \) function
    10440. -
    10441. The \( \chi^2 \) function
    10442. -
    10443. The \( \chi^2 \) function
    10444. -
    10445. The \( \chi^2 \) function
    10446. -
    10447. Fitting an Equation of State for Dense Nuclear Matter
    10448. -
    10449. The code
    10450. -
    10451. Splitting our Data in Training and Test data
    10452. -
    10453. Exercises
    10454. -
    10455. Exercise 1: Setting up various Python environments
    10456. -
    10457. Exercise 2: making your own data and exploring scikit-learn
    10458. -
    10459. Exercise 3: Normalizing our data
    10460. +
    10461. Numpy and arrays
    10462. +
    10463. Matrices in Python
    10464. +
    10465. Meet the Pandas
    10466. +
    10467.    Simple linear regression model using scikit-learn
    10468. +
    10469.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10470. +
    10471.    Organizing our data
    10472. +
    10473.    And what about using neural networks?
    10474. +
    10475. A first summary
    10476. +
    10477. Why Linear Regression (aka Ordinary Least Squares and family)
    10478. +
    10479. Regression analysis, overarching aims
    10480. +
    10481. Regression analysis, overarching aims II
    10482. +
    10483. Examples
    10484. +
    10485. General linear models
    10486. +
    10487. Rewriting the fitting procedure as a linear algebra problem
    10488. +
    10489. Rewriting the fitting procedure as a linear algebra problem, more details
    10490. +
    10491. Generalizing the fitting procedure as a linear algebra problem
    10492. +
    10493. Generalizing the fitting procedure as a linear algebra problem
    10494. +
    10495. Optimizing our parameters
    10496. +
    10497. Our model for the nuclear binding energies
    10498. +
    10499. Optimizing our parameters, more details
    10500. +
    10501. Interpretations and optimizing our parameters
    10502. +
    10503. Interpretations and optimizing our parameters
    10504. +
    10505. Some useful matrix and vector expressions
    10506. +
    10507. Interpretations and optimizing our parameters
    10508. +
    10509. Own code for Ordinary Least Squares
    10510. +
    10511. Adding error analysis and training set up
    10512. +
    10513. The \( \chi^2 \) function
    10514. +
    10515. The \( \chi^2 \) function
    10516. +
    10517. The \( \chi^2 \) function
    10518. +
    10519. The \( \chi^2 \) function
    10520. +
    10521. The \( \chi^2 \) function
    10522. +
    10523. The \( \chi^2 \) function
    10524. +
    10525. Fitting an Equation of State for Dense Nuclear Matter
    10526. +
    10527. The code
    10528. +
    10529. Splitting our Data in Training and Test data
    10530. +
    10531. Exercises
    10532. +
    10533. Exercise 1: Setting up various Python environments
    10534. +
    10535. Exercise 2: making your own data and exploring scikit-learn
    10536. +
    10537. Exercise 3: Split data in test and training data
    10538. @@ -351,25 +335,104 @@ MathJax.Hub.Config({

       

       

       

      -

      The \( \chi^2 \) function

      -
      -
      - -

      The first step here is to approximate the function \( y \) with a first-order polynomial, that is we write

      -$$ -y=y(x) \rightarrow y(x_i) \approx \beta_0+\beta_1 x_i. -$$ +

      Splitting our Data in Training and Test data

      -

      By computing the derivatives of \( \chi^2 \) with respect to \( \beta_0 \) and \( \beta_1 \) show that these are given by

      -$$ -\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_0} = -2\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0, -$$ +

      It is normal in essentially all Machine Learning studies to split the +data in a training set and a test set (sometimes also an additional +validation set). Scikit-Learn has an own function for this. There +is no explicit recipe for how much data should be included as training +data and say test data. An accepted rule of thumb is to use +approximately \( 2/3 \) to \( 4/5 \) of the data as training data. We will +postpone a discussion of this splitting to the end of these notes and +our discussion of the so-called bias-variance tradeoff. Here we +limit ourselves to repeat the above equation of state fitting example +but now splitting the data into a training set and a test set. +

      -

      and

      -$$ -\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_1} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_i\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0. -$$ + + +
      +
      +
      +
      +
      +
      import os
      +import numpy as np
      +import pandas as pd
      +import matplotlib.pyplot as plt
      +from sklearn.model_selection import train_test_split
      +# Where to save the figures and data files
      +PROJECT_ROOT_DIR = "Results"
      +FIGURE_ID = "Results/FigureFiles"
      +DATA_ID = "DataFiles/"
      +
      +if not os.path.exists(PROJECT_ROOT_DIR):
      +    os.mkdir(PROJECT_ROOT_DIR)
      +
      +if not os.path.exists(FIGURE_ID):
      +    os.makedirs(FIGURE_ID)
      +
      +if not os.path.exists(DATA_ID):
      +    os.makedirs(DATA_ID)
      +
      +def image_path(fig_id):
      +    return os.path.join(FIGURE_ID, fig_id)
      +
      +def data_path(dat_id):
      +    return os.path.join(DATA_ID, dat_id)
      +
      +def save_fig(fig_id):
      +    plt.savefig(image_path(fig_id) + ".png", format='png')
      +
      +def R2(y_data, y_model):
      +    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
      +def MSE(y_data,y_model):
      +    n = np.size(y_model)
      +    return np.sum((y_data-y_model)**2)/n
      +
      +infile = open(data_path("EoS.csv"),'r')
      +
      +# Read the EoS data as  csv file and organized into two arrays with density and energies
      +EoS = pd.read_csv(infile, names=('Density', 'Energy'))
      +EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')
      +EoS = EoS.dropna()
      +Energies = EoS['Energy']
      +Density = EoS['Density']
      +#  The design matrix now as function of various polytrops
      +X = np.zeros((len(Density),5))
      +X[:,0] = 1
      +X[:,1] = Density**(2.0/3.0)
      +X[:,2] = Density
      +X[:,3] = Density**(4.0/3.0)
      +X[:,4] = Density**(5.0/3.0)
      +# We split the data in test and training data
      +X_train, X_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)
      +# matrix inversion to find beta
      +beta = np.linalg.inv(X_train.T.dot(X_train)).dot(X_train.T).dot(y_train)
      +# and then make the prediction
      +ytilde = X_train @ beta
      +print("Training R2")
      +print(R2(y_train,ytilde))
      +print("Training MSE")
      +print(MSE(y_train,ytilde))
      +ypredict = X_test @ beta
      +print("Test R2")
      +print(R2(y_test,ypredict))
      +print("Test MSE")
      +print(MSE(y_test,ypredict))
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      @@ -389,10 +452,6 @@ $$
    10539. 60
    10540. 61
    10541. 62
    10542. -
    10543. 63
    10544. -
    10545. 64
    10546. -
    10547. 65
    10548. -
    10549. 66
    10550. »
    10551. diff --git a/doc/pub/week34/html/._week34-bs061.html b/doc/pub/week34/html/._week34-bs061.html index 59e886c67..e49663ca8 100644 --- a/doc/pub/week34/html/._week34-bs061.html +++ b/doc/pub/week34/html/._week34-bs061.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    10552. Installing R, C++, cython or Julia
    10553. Installing R, C++, cython, Numba etc
    10554. Numpy examples and Important Matrix and vector handling packages
    10555. -
    10556. Basic Matrix Features
    10557. -
    10558.    Some famous Matrices
    10559. -
    10560.    More Basic Matrix Features
    10561. -
    10562. Numpy and arrays
    10563. -
    10564. Matrices in Python
    10565. -
    10566. Meet the Pandas
    10567. -
    10568. Friday August 27
    10569. -
    10570.    Simple linear regression model using scikit-learn
    10571. -
    10572.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10573. -
    10574.    Organizing our data
    10575. -
    10576.    Seeing the wood for the trees
    10577. -
    10578.    And what about using neural networks?
    10579. -
    10580. A first summary
    10581. -
    10582. Why Linear Regression (aka Ordinary Least Squares and family)
    10583. -
    10584. Regression analysis, overarching aims
    10585. -
    10586. Regression analysis, overarching aims II
    10587. -
    10588. Examples
    10589. -
    10590. General linear models
    10591. -
    10592. Rewriting the fitting procedure as a linear algebra problem
    10593. -
    10594. Rewriting the fitting procedure as a linear algebra problem, more details
    10595. -
    10596. Generalizing the fitting procedure as a linear algebra problem
    10597. -
    10598. Generalizing the fitting procedure as a linear algebra problem
    10599. -
    10600. Optimizing our parameters
    10601. -
    10602. Our model for the nuclear binding energies
    10603. -
    10604. Optimizing our parameters, more details
    10605. -
    10606. Interpretations and optimizing our parameters
    10607. -
    10608. Interpretations and optimizing our parameters
    10609. -
    10610. Some useful matrix and vector expressions
    10611. -
    10612. Interpretations and optimizing our parameters
    10613. -
    10614. Own code for Ordinary Least Squares
    10615. -
    10616. Adding error analysis and training set up
    10617. -
    10618. The \( \chi^2 \) function
    10619. -
    10620. The \( \chi^2 \) function
    10621. -
    10622. The \( \chi^2 \) function
    10623. -
    10624. The \( \chi^2 \) function
    10625. -
    10626. The \( \chi^2 \) function
    10627. -
    10628. The \( \chi^2 \) function
    10629. -
    10630. Fitting an Equation of State for Dense Nuclear Matter
    10631. -
    10632. The code
    10633. -
    10634. Splitting our Data in Training and Test data
    10635. -
    10636. Exercises
    10637. -
    10638. Exercise 1: Setting up various Python environments
    10639. -
    10640. Exercise 2: making your own data and exploring scikit-learn
    10641. -
    10642. Exercise 3: Normalizing our data
    10643. +
    10644. Numpy and arrays
    10645. +
    10646. Matrices in Python
    10647. +
    10648. Meet the Pandas
    10649. +
    10650.    Simple linear regression model using scikit-learn
    10651. +
    10652.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10653. +
    10654.    Organizing our data
    10655. +
    10656.    And what about using neural networks?
    10657. +
    10658. A first summary
    10659. +
    10660. Why Linear Regression (aka Ordinary Least Squares and family)
    10661. +
    10662. Regression analysis, overarching aims
    10663. +
    10664. Regression analysis, overarching aims II
    10665. +
    10666. Examples
    10667. +
    10668. General linear models
    10669. +
    10670. Rewriting the fitting procedure as a linear algebra problem
    10671. +
    10672. Rewriting the fitting procedure as a linear algebra problem, more details
    10673. +
    10674. Generalizing the fitting procedure as a linear algebra problem
    10675. +
    10676. Generalizing the fitting procedure as a linear algebra problem
    10677. +
    10678. Optimizing our parameters
    10679. +
    10680. Our model for the nuclear binding energies
    10681. +
    10682. Optimizing our parameters, more details
    10683. +
    10684. Interpretations and optimizing our parameters
    10685. +
    10686. Interpretations and optimizing our parameters
    10687. +
    10688. Some useful matrix and vector expressions
    10689. +
    10690. Interpretations and optimizing our parameters
    10691. +
    10692. Own code for Ordinary Least Squares
    10693. +
    10694. Adding error analysis and training set up
    10695. +
    10696. The \( \chi^2 \) function
    10697. +
    10698. The \( \chi^2 \) function
    10699. +
    10700. The \( \chi^2 \) function
    10701. +
    10702. The \( \chi^2 \) function
    10703. +
    10704. The \( \chi^2 \) function
    10705. +
    10706. The \( \chi^2 \) function
    10707. +
    10708. Fitting an Equation of State for Dense Nuclear Matter
    10709. +
    10710. The code
    10711. +
    10712. Splitting our Data in Training and Test data
    10713. +
    10714. Exercises
    10715. +
    10716. Exercise 1: Setting up various Python environments
    10717. +
    10718. Exercise 2: making your own data and exploring scikit-learn
    10719. +
    10720. Exercise 3: Split data in test and training data
    10721. @@ -351,58 +335,210 @@ MathJax.Hub.Config({

       

       

       

      -

      The \( \chi^2 \) function

      -
      -
      - +

      Exercises

      -

      For a linear fit (a first-order polynomial) we don't need to invert a matrix!! -Defining +

      Here are three possible exercises for week 34

      + + +

      Exercise 1: Setting up various Python environments

      + +

      The first exercise here is of a mere technical art. We want you to have

      +
        +
      • git as a version control software and to establish a user account on a provider like GitHub. Other providers like GitLab etc are equally fine. You can also use the University of Oslo GitHub facilities.
      • +
      • Install various Python packages
      • +
      +

      We will make extensive use of Python as programming language and its +myriad of available libraries. You will find +IPython/Jupyter notebooks invaluable in your work. You can run R +codes in the Jupyter/IPython notebooks, with the immediate benefit of +visualizing your data. You can also use compiled languages like C++, +Rust, Fortran etc if you prefer. The focus in these lectures will be +on Python.

      -$$ -\gamma = \sum_{i=0}^{n-1}\frac{1}{\sigma_i^2}, -$$ - -$$ -\gamma_x = \sum_{i=0}^{n-1}\frac{x_{i}}{\sigma_i^2}, -$$ - - -$$ -\gamma_y = \sum_{i=0}^{n-1}\left(\frac{y_i}{\sigma_i^2}\right), -$$ - - -$$ -\gamma_{xx} = \sum_{i=0}^{n-1}\frac{x_ix_{i}}{\sigma_i^2}, -$$ - - -$$ -\gamma_{xy} = \sum_{i=0}^{n-1}\frac{y_ix_{i}}{\sigma_i^2}, -$$ - -

      we obtain

      - -$$ -\beta_0 = \frac{\gamma_{xx}\gamma_y-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}, -$$ - - -$$ -\beta_1 = \frac{\gamma_{xy}\gamma-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}. -$$ - -

      This approach (different linear and non-linear regression) suffers -often from both being underdetermined and overdetermined in the -unknown coefficients \( \beta_i \). A better approach is to use the -Singular Value Decomposition (SVD) method discussed next week. +

      If you have Python installed (we recommend Python3) and you feel +pretty familiar with installing different packages, we recommend that +you install the following Python packages via pip as

      + +
        +
      1. pip install numpy scipy matplotlib ipython scikit-learn sympy pandas pillow
      2. +
      +

      For Tensorflow, we recommend following the instructions in the text of +Aurelien Geron, Hands‑On Machine Learning with Scikit‑Learn and TensorFlow, O'Reilly +

      + +

      We will come back to tensorflow later.

      + +

      For Python3, replace pip with pip3.

      + +

      For OSX users we recommend, after having installed Xcode, to +install brew. Brew allows for a seamless installation of additional +software via for example +

      + +
        +
      1. brew install python3
      2. +
      +

      For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution, +you can use pip as well and simply install Python as +

      + +
        +
      1. sudo apt-get install python3 (or python for Python2.7)
      2. +
      +

      If you don't want to perform these operations separately and venture +into the hassle of exploring how to set up dependencies and paths, we +recommend two widely used distrubutions which set up all relevant +dependencies for Python, namely +

      + + +

      which is an open source +distribution of the Python and R programming languages for large-scale +data processing, predictive analytics, and scientific computing, that +aims to simplify package management and deployment. Package versions +are managed by the package management system conda. +

      + + +

      is a Python +distribution for scientific and analytic computing distribution and +analysis environment, available for free and under a commercial +license. +

      + +

      We recommend using Anaconda if you are not too familiar with setting paths in a terminal environment.

      + + + + +

      Exercise 2: making your own data and exploring scikit-learn

      + +

      We will generate our own dataset for a function \( y(x) \) where \( x \in [0,1] \) and defined by random numbers computed with the uniform distribution. The function \( y \) is a quadratic polynomial in \( x \) with added stochastic noise according to the normal distribution \( \cal {N}(0,1) \). +The following simple Python instructions define our \( x \) and \( y \) values (with 100 data points). +

      + + +
      +
      +
      +
      +
      +
      x = np.random.rand(100,1)
      +y = 2.0+5*x*x+0.1*np.random.randn(100,1)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
        +
      1. Write your own code (following the examples under the regression notes) for computing the parametrization of the data set fitting a second-order polynomial.
      2. +
      3. Use thereafter scikit-learn (see again the examples in the regression slides) and compare with your own code.
      4. +
      5. Using scikit-learn, compute also the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error defined as
      6. +
      +$$ MSE(\boldsymbol{y},\boldsymbol{\tilde{y}}) = \frac{1}{n} +\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, +$$ + +

      and the \( R^2 \) score function. +If \( \tilde{\boldsymbol{y}}_i \) is the predicted value of the \( i-th \) sample and \( y_i \) is the corresponding true value, then the score \( R^2 \) is defined as +

      +$$ +R^2(\boldsymbol{y}, \tilde{\boldsymbol{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, +$$ + +

      where we have defined the mean value of \( \boldsymbol{y} \) as

      +$$ +\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. +$$ + +

      You can use the functionality included in scikit-learn. If you feel for it, you can use your own program and define functions which compute the above two functions. +Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits. +

      + + + + +

      Exercise 3: Split data in test and training data

      + +

      In this exercise we want you to to compute the MSE for the training +data and the test data as function of the complexity of a polynomial, +that is the degree of a given polynomial. +

      + +

      The aim is to reproduce Figure 2.11 of Hastie et al.

      + +

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points. You should try to vary \( n \) in your analysis.

      + + +
      +
      +
      +
      +
      +
      np.random.seed()
      +n = 100
      +# Make data set.
      +x = np.linspace(-3, 3, n).reshape(-1, 1)
      +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      +
      + +

      where \( y \) is the function we want to fit with a given polynomial.

      + + +

      +a) +Write a first code which sets up a design matrix \( X \) defined by a fifth-order polynomial and split your data set in training and test data. +

      + + + + +

      +b) +Perform an ordinary least squares fitting and compute the means squared error for the training data and the test data. +

      + + + + +

      +c) +Add now a model which allows you to make polynomials up to degree \( 15 \). Perform a standard OLS fitting of the training data and compute the MSE for the training and test data and plot both test and training data MSE as functions of the polynomial degree. Compare what you see with Figure 2.11 of Hastie et al. Comment your results. For which polynomial degree do you find an optimal MSE (smallest value)? +

      + + + +

        @@ -418,11 +554,6 @@ Singular Value Decomposition (SVD) method discussed next week.
      • 60
      • 61
      • 62
      • -
      • 63
      • -
      • 64
      • -
      • 65
      • -
      • 66
      • -
      • »
      diff --git a/doc/pub/week34/html/week34-bs.html b/doc/pub/week34/html/week34-bs.html index 9aaab6cf1..19b4a2d84 100644 --- a/doc/pub/week34/html/week34-bs.html +++ b/doc/pub/week34/html/week34-bs.html @@ -109,16 +109,9 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'numpy-examples-and-important-matrix-and-vector-handling-packages'), - ('Basic Matrix Features', 2, None, 'basic-matrix-features'), - ('Some famous Matrices', 3, None, 'some-famous-matrices'), - ('More Basic Matrix Features', - 3, - None, - 'more-basic-matrix-features'), ('Numpy and arrays', 2, None, 'numpy-and-arrays'), ('Matrices in Python', 2, None, 'matrices-in-python'), ('Meet the Pandas', 2, None, 'meet-the-pandas'), - ('Friday August 27', 2, None, 'friday-august-27'), ('Simple linear regression model using _scikit-learn_', 3, None, @@ -129,10 +122,6 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d None, 'to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies'), ('Organizing our data', 3, None, 'organizing-our-data'), - ('Seeing the wood for the trees', - 3, - None, - 'seeing-the-wood-for-the-trees'), ('And what about using neural networks?', 3, None, @@ -229,10 +218,10 @@ doconce format html week34.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'exercise-2-making-your-own-data-and-exploring-scikit-learn'), - ('Exercise 3: Normalizing our data', + ('Exercise 3: Split data in test and training data', 2, None, - 'exercise-3-normalizing-our-data')]} + 'exercise-3-split-data-in-test-and-training-data')]} end of tocinfo --> @@ -296,50 +285,45 @@ MathJax.Hub.Config({
    10722. Installing R, C++, cython or Julia
    10723. Installing R, C++, cython, Numba etc
    10724. Numpy examples and Important Matrix and vector handling packages
    10725. -
    10726. Basic Matrix Features
    10727. -
    10728.    Some famous Matrices
    10729. -
    10730.    More Basic Matrix Features
    10731. -
    10732. Numpy and arrays
    10733. -
    10734. Matrices in Python
    10735. -
    10736. Meet the Pandas
    10737. -
    10738. Friday August 27
    10739. -
    10740.    Simple linear regression model using scikit-learn
    10741. -
    10742.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10743. -
    10744.    Organizing our data
    10745. -
    10746.    Seeing the wood for the trees
    10747. -
    10748.    And what about using neural networks?
    10749. -
    10750. A first summary
    10751. -
    10752. Why Linear Regression (aka Ordinary Least Squares and family)
    10753. -
    10754. Regression analysis, overarching aims
    10755. -
    10756. Regression analysis, overarching aims II
    10757. -
    10758. Examples
    10759. -
    10760. General linear models
    10761. -
    10762. Rewriting the fitting procedure as a linear algebra problem
    10763. -
    10764. Rewriting the fitting procedure as a linear algebra problem, more details
    10765. -
    10766. Generalizing the fitting procedure as a linear algebra problem
    10767. -
    10768. Generalizing the fitting procedure as a linear algebra problem
    10769. -
    10770. Optimizing our parameters
    10771. -
    10772. Our model for the nuclear binding energies
    10773. -
    10774. Optimizing our parameters, more details
    10775. -
    10776. Interpretations and optimizing our parameters
    10777. -
    10778. Interpretations and optimizing our parameters
    10779. -
    10780. Some useful matrix and vector expressions
    10781. -
    10782. Interpretations and optimizing our parameters
    10783. -
    10784. Own code for Ordinary Least Squares
    10785. -
    10786. Adding error analysis and training set up
    10787. -
    10788. The \( \chi^2 \) function
    10789. -
    10790. The \( \chi^2 \) function
    10791. -
    10792. The \( \chi^2 \) function
    10793. -
    10794. The \( \chi^2 \) function
    10795. -
    10796. The \( \chi^2 \) function
    10797. -
    10798. The \( \chi^2 \) function
    10799. -
    10800. Fitting an Equation of State for Dense Nuclear Matter
    10801. -
    10802. The code
    10803. -
    10804. Splitting our Data in Training and Test data
    10805. -
    10806. Exercises
    10807. -
    10808. Exercise 1: Setting up various Python environments
    10809. -
    10810. Exercise 2: making your own data and exploring scikit-learn
    10811. -
    10812. Exercise 3: Normalizing our data
    10813. +
    10814. Numpy and arrays
    10815. +
    10816. Matrices in Python
    10817. +
    10818. Meet the Pandas
    10819. +
    10820.    Simple linear regression model using scikit-learn
    10821. +
    10822.    To our real data: nuclear binding energies. Brief reminder on masses and binding energies
    10823. +
    10824.    Organizing our data
    10825. +
    10826.    And what about using neural networks?
    10827. +
    10828. A first summary
    10829. +
    10830. Why Linear Regression (aka Ordinary Least Squares and family)
    10831. +
    10832. Regression analysis, overarching aims
    10833. +
    10834. Regression analysis, overarching aims II
    10835. +
    10836. Examples
    10837. +
    10838. General linear models
    10839. +
    10840. Rewriting the fitting procedure as a linear algebra problem
    10841. +
    10842. Rewriting the fitting procedure as a linear algebra problem, more details
    10843. +
    10844. Generalizing the fitting procedure as a linear algebra problem
    10845. +
    10846. Generalizing the fitting procedure as a linear algebra problem
    10847. +
    10848. Optimizing our parameters
    10849. +
    10850. Our model for the nuclear binding energies
    10851. +
    10852. Optimizing our parameters, more details
    10853. +
    10854. Interpretations and optimizing our parameters
    10855. +
    10856. Interpretations and optimizing our parameters
    10857. +
    10858. Some useful matrix and vector expressions
    10859. +
    10860. Interpretations and optimizing our parameters
    10861. +
    10862. Own code for Ordinary Least Squares
    10863. +
    10864. Adding error analysis and training set up
    10865. +
    10866. The \( \chi^2 \) function
    10867. +
    10868. The \( \chi^2 \) function
    10869. +
    10870. The \( \chi^2 \) function
    10871. +
    10872. The \( \chi^2 \) function
    10873. +
    10874. The \( \chi^2 \) function
    10875. +
    10876. The \( \chi^2 \) function
    10877. +
    10878. Fitting an Equation of State for Dense Nuclear Matter
    10879. +
    10880. The code
    10881. +
    10882. Splitting our Data in Training and Test data
    10883. +
    10884. Exercises
    10885. +
    10886. Exercise 1: Setting up various Python environments
    10887. +
    10888. Exercise 2: making your own data and exploring scikit-learn
    10889. +
    10890. Exercise 3: Split data in test and training data
    10891. @@ -394,7 +378,7 @@ MathJax.Hub.Config({
    10892. 9
    10893. 10
    10894. ...
    10895. -
    10896. 66
    10897. +
    10898. 62
    10899. »
    10900. diff --git a/doc/pub/week34/html/week34-reveal.html b/doc/pub/week34/html/week34-reveal.html index fca2cf221..4b6297059 100644 --- a/doc/pub/week34/html/week34-reveal.html +++ b/doc/pub/week34/html/week34-reveal.html @@ -199,7 +199,7 @@ MathJax.Hub.Config({

      1. The sessions on Tuesdays and Wednesdays last four hours for each group (four groups in total) and will include lectures in a flipped mode (promoting active learning) and work on exercices and projects.
      2. -

      3. The sessions will begin with lectures, discussions, questions and answers about the material to be covered every week.
      4. +

      5. The sessions will begin with lectures, discussions, questions and answers about the material to be covered every week. Videos and teaching material will be announced in due time.
      6. There are four groups:
        • @@ -579,9 +579,9 @@ topics and tools as well as showing the power of various Python libraries for machine learning and statistical data analysis.

          -

          Here, we will mainly focus on two +

          Although we have projects where you write your own codes, we will also focus on two specific Python packages for Machine Learning, Scikit-Learn and -Tensorflow (see below for links etc). Moreover, the examples we +Tensorflow with Keras (see below for links etc). Moreover, the examples we introduce will serve as inputs to many of our discussions later, as well as allowing you to set up models and produce your own data and get started with programming. @@ -901,8 +901,8 @@ use Python during our lectures and in various projects and exercises. Those of you already familiar with R should feel free to continue using R, keeping however an eye on the parallel Python set ups. Similarly, if you are a -Python afecionado, feel free to explore R as well. Jupyter/Ipython -notebook allows you to run R codes interactively in your +Python afecionado, feel free to explore R as well. Jupyter(Julia, Python and R) /Ipython +notebook allows you to run R codes and Julia codes interactively in your browser. The software library R is really tailored for statistical data analysis and allows for an easy usage of the tools and algorithms we will discuss in these lectures. @@ -960,10 +960,7 @@ further processing. For example, convert to latex as

          And to add more versatility, the Python package SymPy is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python.

          -

          Finally, if you wish to use the light mark-up language -doconce you can convert a standard ascii text file into various HTML -formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using doconce. -

          +

          Finally, we recommend strongly using Autograd or JAX for automatic differentiation.

          @@ -985,103 +982,6 @@ developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly he
        -
        -

        Basic Matrix Features

        - -
        -Matrix properties reminder -

        -

         
        -$$ - \mathbf{A} = - \begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\ - a_{21} & a_{22} & a_{23} & a_{24} \\ - a_{31} & a_{32} & a_{33} & a_{34} \\ - a_{41} & a_{42} & a_{43} & a_{44} - \end{bmatrix}\qquad -\mathbf{I} = - \begin{bmatrix} 1 & 0 & 0 & 0 \\ - 0 & 1 & 0 & 0 \\ - 0 & 0 & 1 & 0 \\ - 0 & 0 & 0 & 1 - \end{bmatrix} -$$ -

         
        - -

        The inverse of a matrix is defined by

        - -

         
        -$$ -\mathbf{A}^{-1} \cdot \mathbf{A} = I -$$ -

         
        - - - - - - - - - - - - - -
        Relations Name matrix elements
        \( A=A^{T} \) symmetric \( a_{ij}=a_{ji} \)
        \( A=\left (A^{T}\right )^{-1} \) real orthogonal \( \sum_k a_{ik}a_{jk}=\sum_k a_{ki} a_{kj}=\delta_{ij} \)
        \( A=A^* \) real matrix \( a_{ij}=a_{ij}^* \)
        \( A=A^{\dagger} \) hermitian \( a_{ij}=a_{ji}^* \)
        \( A=\left(A^{\dagger}\right )^{-1} \) unitary \( \sum_k a_{ik}a_{jk}^*=\sum_k a_{ki}^* a_{kj}=\delta_{ij} \)
        -

        -
        - -
        -

        Some famous Matrices

        - -
          - -

        • Diagonal if \( a_{ij}=0 \) for \( i\ne j \)
        • - -

        • Upper triangular if \( a_{ij}=0 \) for \( i>j \)
        • - -

        • Lower triangular if \( a_{ij}=0 \) for \( i < j \)
        • - -

        • Upper Hessenberg if \( a_{ij}=0 \) for \( i>j+1 \)
        • - -

        • Lower Hessenberg if \( a_{ij}=0 \) for \( i < j+1 \)
        • - -

        • Tridiagonal if \( a_{ij}=0 \) for \( |i -j|>1 \)
        • - -

        • Lower banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i>j+p \)
        • - -

        • Upper banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i < j+p \)
        • - -

        • Banded, block upper triangular, block lower triangular....
        • -
        -
        - -
        -

        More Basic Matrix Features

        - -
        -Some Equivalent Statements -

        -

        For an \( N\times N \) matrix \( \mathbf{A} \) the following properties are all equivalent

        - -
          - -

        • If the inverse of \( \mathbf{A} \) exists, \( \mathbf{A} \) is nonsingular.
        • - -

        • The equation \( \mathbf{Ax}=0 \) implies \( \mathbf{x}=0 \).
        • - -

        • The rows of \( \mathbf{A} \) form a basis of \( R^N \).
        • - -

        • The columns of \( \mathbf{A} \) form a basis of \( R^N \).
        • - -

        • \( \mathbf{A} \) is a product of elementary matrices.
        • - -

        • \( 0 \) is not eigenvalue of \( \mathbf{A} \).
        • -
        -
        -
        -

        Numpy and arrays

        Numpy provides an easy way to handle arrays in Python. The standard way to import this library is as

        @@ -1837,14 +1737,6 @@ For multidimensional arrays, we recommend strongly Friday August 27 - -

        "Video of Lecture August 27, 2021":"https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h21/forelesningsvideoer/LectureThursdayAugust27.mp4?vrtx=view-as-webpage

        - -

        Video of Lecture from fall 2020 and Handwritten notes

        -
        -

        Simple linear regression model using scikit-learn

        @@ -2197,62 +2089,8 @@ $$

        Here \( \boldsymbol{a}=\boldsymbol{y} - \boldsymbol{\tilde{y}} \).

        We will discuss in more detail these and other functions in the -various lectures. We conclude this part with another example. Instead -of a linear \( x \)-dependence we study now a cubic polynomial and use the -polynomial regression analysis tools of scikit-learn. +various lectures and lab sessions.

        - - - -
        -
        -
        -
        -
        -
        import matplotlib.pyplot as plt
        -import numpy as np
        -import random
        -from sklearn.linear_model import Ridge
        -from sklearn.preprocessing import PolynomialFeatures
        -from sklearn.pipeline import make_pipeline
        -from sklearn.linear_model import LinearRegression
        -
        -x=np.linspace(0.02,0.98,200)
        -noise = np.asarray(random.sample((range(200)),200))
        -y=x**3*noise
        -yn=x**3*100
        -poly3 = PolynomialFeatures(degree=3)
        -X = poly3.fit_transform(x[:,np.newaxis])
        -clf3 = LinearRegression()
        -clf3.fit(X,y)
        -
        -Xplot=poly3.fit_transform(x[:,np.newaxis])
        -poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')
        -plt.plot(x,yn, color='red', label="True Cubic")
        -plt.scatter(x, y, label='Data', color='orange', s=15)
        -plt.legend()
        -plt.show()
        -
        -def error(a):
        -    for i in y:
        -        err=(y-yn)/yn
        -    return abs(np.sum(err))/len(err)
        -
        -print (error(y))
        -
        -
        -
        -
        -
        -
        -
        -
        -
        -
        -
        -
        -
        -

        To our real data: nuclear binding energies. Brief reminder on masses and binding energies

        Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding @@ -2669,60 +2507,6 @@ plt.show()

      -

      Seeing the wood for the trees

      - -

      As a teaser, let us now see how we can do this with decision trees using scikit-learn. Later we will switch to so-called random forests!

      - - - -
      -
      -
      -
      -
      -
      #Decision Tree Regression
      -from sklearn.tree import DecisionTreeRegressor
      -regr_1=DecisionTreeRegressor(max_depth=5)
      -regr_2=DecisionTreeRegressor(max_depth=7)
      -regr_3=DecisionTreeRegressor(max_depth=9)
      -regr_1.fit(X, Energies)
      -regr_2.fit(X, Energies)
      -regr_3.fit(X, Energies)
      -
      -
      -y_1 = regr_1.predict(X)
      -y_2 = regr_2.predict(X)
      -y_3=regr_3.predict(X)
      -Masses['Eapprox'] = y_3
      -# Plot the results
      -plt.figure()
      -plt.plot(A, Energies, color="blue", label="Data", linewidth=2)
      -plt.plot(A, y_1, color="red", label="max_depth=5", linewidth=2)
      -plt.plot(A, y_2, color="green", label="max_depth=7", linewidth=2)
      -plt.plot(A, y_3, color="m", label="max_depth=9", linewidth=2)
      -
      -plt.xlabel("$A$")
      -plt.ylabel("$E$[MeV]")
      -plt.title("Decision Tree Regression")
      -plt.legend()
      -save_fig("Masses2016Trees")
      -plt.show()
      -print(Masses)
      -print(np.mean( (Energies-y_1)**2))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      And what about using neural networks?

      The seaborn package allows us to visualize data in an efficient way. Note that we use scikit-learn's multi-layer perceptron (or feed forward neural network) functionality. @@ -2867,7 +2651,7 @@ the linear regression model where \( \boldsymbol{\beta} = [\beta_0, \ld

      -

      In order to understand the relation among the predictors \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), +

      In order to understand the relation among the predictors (or features or properties) \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), consider the model we discussed for describing nuclear binding energies.

      @@ -3331,33 +3115,9 @@ allow for the usage of direct linear algebra methods such as LU decomposi

      Some useful matrix and vector expressions

      -

      The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and -matrices as upper case boldfaced letters. -

      +

      See the handwritten notes at https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/2022/NotesExercise5Week452022.pdf

      -

       
      -$$ -\frac{\partial (\boldsymbol{b}^T\boldsymbol{a})}{\partial \boldsymbol{a}} = \boldsymbol{b}, -$$ -

       
      - -

       
      -$$ -\frac{\partial (\boldsymbol{a}^T\boldsymbol{A}\boldsymbol{a})}{\partial \boldsymbol{a}} = (\boldsymbol{A}+\boldsymbol{A}^T)\boldsymbol{a}, -$$ -

       
      - -

       
      -$$ -\frac{\partial tr(\boldsymbol{B}\boldsymbol{A})}{\partial \boldsymbol{A}} = \boldsymbol{B}^T, -$$ -

       
      - -

       
      -$$ -\frac{\partial \log{\vert\boldsymbol{A}\vert}}{\partial \boldsymbol{A}} = (\boldsymbol{A}^{-1})^T. -$$ -

       
      +

      These notes will be discussed during one of the lectures.

      @@ -4057,7 +3817,8 @@ ypredict = X_test @ beta

      Exercises

      -

      Here are three possible exercises for weeks 34 and 35.

      + +

      Here are three possible exercises for week 34

      Exercise 1: Setting up various Python environments

      @@ -4206,181 +3967,19 @@ $$ Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits.

      - -

      -Solution. -The code here is an example of where we define our own design matrix and fit parameters \( \beta \). -

      - - -
      -
      -
      -
      -
      -
      import os
      -import numpy as np
      -import pandas as pd
      -import matplotlib.pyplot as plt
      -from sklearn.model_selection import train_test_split
      -
      -def save_fig(fig_id):
      -    plt.savefig(image_path(fig_id) + ".png", format='png')
      -
      -def R2(y_data, y_model):
      -    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
      -def MSE(y_data,y_model):
      -    n = np.size(y_model)
      -    return np.sum((y_data-y_model)**2)/n
      -
      -x = np.random.rand(100)
      -y = 2.0+5*x*x+0.1*np.random.randn(100)
      -
      -
      -#  The design matrix now as function of a given polynomial
      -X = np.zeros((len(x),3))
      -X[:,0] = 1.0
      -X[:,1] = x
      -X[:,2] = x**2
      -# We split the data in test and training data
      -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
      -# matrix inversion to find beta
      -beta = np.linalg.inv(X_train.T @ X_train) @ X_train.T @ y_train
      -print(beta)
      -# and then make the prediction
      -ytilde = X_train @ beta
      -print("Training R2")
      -print(R2(y_train,ytilde))
      -print("Training MSE")
      -print(MSE(y_train,ytilde))
      -ypredict = X_test @ beta
      -print("Test R2")
      -print(R2(y_test,ypredict))
      -print("Test MSE")
      -print(MSE(y_test,ypredict))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - - - - -

      Exercise 3: Normalizing our data

      - -

      A much used approach before starting to train the data is to preprocess our -data. Normally the data may need a rescaling and/or may be sensitive -to extreme values. Scaling the data renders our inputs much more -suitable for the algorithms we want to employ. -

      - -

      Scikit-Learn has several functions which allow us to rescale the -data, normally resulting in much better results in terms of various -accuracy scores. The StandardScaler function in Scikit-Learn -ensures that for each feature/predictor we study the mean value is -zero and the variance is one (every column in the design/feature -matrix). This scaling has the drawback that it does not ensure that -we have a particular maximum or minimum in our data set. Another -function included in Scikit-Learn is the MinMaxScaler which -ensures that all features are exactly between \( 0 \) and \( 1 \). The -

      - -

      The Normalizer scales each data -point such that the feature vector has a euclidean length of one. In other words, it -projects a data point on the circle (or sphere in the case of higher dimensions) with a -radius of 1. This means every data point is scaled by a different number (by the -inverse of it’s length). -This normalization is often used when only the direction (or angle) of the data matters, -not the length of the feature vector. -

      - -

      The RobustScaler works similarly to the StandardScaler in that it -ensures statistical properties for each feature that guarantee that -they are on the same scale. However, the RobustScaler uses the median -and quartiles, instead of mean and variance. This makes the -RobustScaler ignore data points that are very different from the rest -(like measurement errors). These odd data points are also called -outliers, and might often lead to trouble for other scaling -techniques. -

      - -

      It also common to split the data in a training set and a testing set. A typical split is to use \( 80\% \) of the data for training and the rest -for testing. This can be done as follows with our design matrix \( \boldsymbol{X} \) and data \( \boldsymbol{y} \) (remember to import scikit-learn) -

      - - -
      -
      -
      -
      -
      -
      # split in training and test data
      -X_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.2)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Then we can use the standard scaler to scale our data as

      - - -
      -
      -
      -
      -
      -
      scaler = StandardScaler()
      -scaler.fit(X_train)
      -X_train_scaled = scaler.transform(X_train)
      -X_test_scaled = scaler.transform(X_test)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      +

      Exercise 3: Split data in test and training data

      In this exercise we want you to to compute the MSE for the training data and the test data as function of the complexity of a polynomial, -that is the degree of a given polynomial. We want you also to compute the \( R2 \) score as function of the complexity of the model for both training data and test data. You should also run the calculation with and without scaling. +that is the degree of a given polynomial.

      -

      One of -the aims is to reproduce Figure 2.11 of Hastie et al. -

      +

      The aim is to reproduce Figure 2.11 of Hastie et al.

      -

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points.

      +

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points. You should try to vary \( n \) in your analysis.

      @@ -4390,7 +3989,6 @@ the aims is to reproduce Figure 2.11 of
      np.random.seed()
       n = 100
      -maxdegree = 14
       # Make data set.
       x = np.linspace(-3, 3, n).reshape(-1, 1)
       y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
      @@ -4414,7 +4012,7 @@ y = np.exp(-x**2) + SymPy is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS)  and is entirely written in Python. 

      -

      Finally, if you wish to use the light mark-up language -doconce you can convert a standard ascii text file into various HTML -formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using doconce. -

      +

      Finally, we recommend strongly using Autograd or JAX for automatic differentiation.











      Numpy examples and Important Matrix and vector handling packages

      @@ -1011,82 +997,6 @@ developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly he
    10901. LAPACK:package for solving symmetric, unsymmetric and generalized eigenvalue problems. From LAPACK's website http://www.netlib.org it is possible to download for free all source codes from this library. Both C/C++ and Fortran versions are available.
    10902. BLAS (I, II and III): (Basic Linear Algebra Subprograms) are routines that provide standard building blocks for performing basic vector and matrix operations. Blas I is vector operations, II vector-matrix operations and III matrix-matrix operations. Highly parallelized and efficient codes, all available for download from http://www.netlib.org.
    10903. -









      -

      Basic Matrix Features

      - -
      -Matrix properties reminder -

      -$$ - \mathbf{A} = - \begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\ - a_{21} & a_{22} & a_{23} & a_{24} \\ - a_{31} & a_{32} & a_{33} & a_{34} \\ - a_{41} & a_{42} & a_{43} & a_{44} - \end{bmatrix}\qquad -\mathbf{I} = - \begin{bmatrix} 1 & 0 & 0 & 0 \\ - 0 & 1 & 0 & 0 \\ - 0 & 0 & 1 & 0 \\ - 0 & 0 & 0 & 1 - \end{bmatrix} -$$ - -

      The inverse of a matrix is defined by

      - -$$ -\mathbf{A}^{-1} \cdot \mathbf{A} = I -$$ - - - - - - - - - - - - - -
      Relations Name matrix elements
      \( A=A^{T} \) symmetric \( a_{ij}=a_{ji} \)
      \( A=\left (A^{T}\right )^{-1} \) real orthogonal \( \sum_k a_{ik}a_{jk}=\sum_k a_{ki} a_{kj}=\delta_{ij} \)
      \( A=A^* \) real matrix \( a_{ij}=a_{ij}^* \)
      \( A=A^{\dagger} \) hermitian \( a_{ij}=a_{ji}^* \)
      \( A=\left(A^{\dagger}\right )^{-1} \) unitary \( \sum_k a_{ik}a_{jk}^*=\sum_k a_{ki}^* a_{kj}=\delta_{ij} \)
      -
      - - -









      -

      Some famous Matrices

      - -
        -
      • Diagonal if \( a_{ij}=0 \) for \( i\ne j \)
      • -
      • Upper triangular if \( a_{ij}=0 \) for \( i>j \)
      • -
      • Lower triangular if \( a_{ij}=0 \) for \( i < j \)
      • -
      • Upper Hessenberg if \( a_{ij}=0 \) for \( i>j+1 \)
      • -
      • Lower Hessenberg if \( a_{ij}=0 \) for \( i < j+1 \)
      • -
      • Tridiagonal if \( a_{ij}=0 \) for \( |i -j|>1 \)
      • -
      • Lower banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i>j+p \)
      • -
      • Upper banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i < j+p \)
      • -
      • Banded, block upper triangular, block lower triangular....
      • -
      -









      -

      More Basic Matrix Features

      - -
      -Some Equivalent Statements -

      -

      For an \( N\times N \) matrix \( \mathbf{A} \) the following properties are all equivalent

      - -
        -
      • If the inverse of \( \mathbf{A} \) exists, \( \mathbf{A} \) is nonsingular.
      • -
      • The equation \( \mathbf{Ax}=0 \) implies \( \mathbf{x}=0 \).
      • -
      • The rows of \( \mathbf{A} \) form a basis of \( R^N \).
      • -
      • The columns of \( \mathbf{A} \) form a basis of \( R^N \).
      • -
      • \( \mathbf{A} \) is a product of elementary matrices.
      • -
      • \( 0 \) is not eigenvalue of \( \mathbf{A} \).
      • -
      -
      - -









      Numpy and arrays

      Numpy provides an easy way to handle arrays in Python. The standard way to import this library is as

      @@ -1835,13 +1745,6 @@ As we will see below it leads also to a very concice code close to the mathemati For multidimensional arrays, we recommend strongly xarray. xarray has much of the same flexibility as pandas, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both pandas and xarray.

      -









      -

      Friday August 27

      - -

      "Video of Lecture August 27, 2021":"https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h21/forelesningsvideoer/LectureThursdayAugust27.mp4?vrtx=view-as-webpage

      - -

      Video of Lecture from fall 2020 and Handwritten notes

      -









      Simple linear regression model using scikit-learn

      @@ -2174,62 +2077,8 @@ $$

      Here \( \boldsymbol{a}=\boldsymbol{y} - \boldsymbol{\tilde{y}} \).

      We will discuss in more detail these and other functions in the -various lectures. We conclude this part with another example. Instead -of a linear \( x \)-dependence we study now a cubic polynomial and use the -polynomial regression analysis tools of scikit-learn. +various lectures and lab sessions.

      - - - -
      -
      -
      -
      -
      -
      import matplotlib.pyplot as plt
      -import numpy as np
      -import random
      -from sklearn.linear_model import Ridge
      -from sklearn.preprocessing import PolynomialFeatures
      -from sklearn.pipeline import make_pipeline
      -from sklearn.linear_model import LinearRegression
      -
      -x=np.linspace(0.02,0.98,200)
      -noise = np.asarray(random.sample((range(200)),200))
      -y=x**3*noise
      -yn=x**3*100
      -poly3 = PolynomialFeatures(degree=3)
      -X = poly3.fit_transform(x[:,np.newaxis])
      -clf3 = LinearRegression()
      -clf3.fit(X,y)
      -
      -Xplot=poly3.fit_transform(x[:,np.newaxis])
      -poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')
      -plt.plot(x,yn, color='red', label="True Cubic")
      -plt.scatter(x, y, label='Data', color='orange', s=15)
      -plt.legend()
      -plt.show()
      -
      -def error(a):
      -    for i in y:
      -        err=(y-yn)/yn
      -    return abs(np.sum(err))/len(err)
      -
      -print (error(y))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      To our real data: nuclear binding energies. Brief reminder on masses and binding energies

      Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding @@ -2630,60 +2479,6 @@ plt.show()

      -

      Seeing the wood for the trees

      - -

      As a teaser, let us now see how we can do this with decision trees using scikit-learn. Later we will switch to so-called random forests!

      - - - -
      -
      -
      -
      -
      -
      #Decision Tree Regression
      -from sklearn.tree import DecisionTreeRegressor
      -regr_1=DecisionTreeRegressor(max_depth=5)
      -regr_2=DecisionTreeRegressor(max_depth=7)
      -regr_3=DecisionTreeRegressor(max_depth=9)
      -regr_1.fit(X, Energies)
      -regr_2.fit(X, Energies)
      -regr_3.fit(X, Energies)
      -
      -
      -y_1 = regr_1.predict(X)
      -y_2 = regr_2.predict(X)
      -y_3=regr_3.predict(X)
      -Masses['Eapprox'] = y_3
      -# Plot the results
      -plt.figure()
      -plt.plot(A, Energies, color="blue", label="Data", linewidth=2)
      -plt.plot(A, y_1, color="red", label="max_depth=5", linewidth=2)
      -plt.plot(A, y_2, color="green", label="max_depth=7", linewidth=2)
      -plt.plot(A, y_3, color="m", label="max_depth=9", linewidth=2)
      -
      -plt.xlabel("$A$")
      -plt.ylabel("$E$[MeV]")
      -plt.title("Decision Tree Regression")
      -plt.legend()
      -save_fig("Masses2016Trees")
      -plt.show()
      -print(Masses)
      -print(np.mean( (Energies-y_1)**2))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      And what about using neural networks?

      The seaborn package allows us to visualize data in an efficient way. Note that we use scikit-learn's multi-layer perceptron (or feed forward neural network) functionality. @@ -2824,7 +2619,7 @@ the linear regression model where \( \boldsymbol{\beta} = [\beta_0, \ld

      -

      In order to understand the relation among the predictors \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), +

      In order to understand the relation among the predictors (or features or properties) \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), consider the model we discussed for describing nuclear binding energies.

      @@ -3235,25 +3030,9 @@ allow for the usage of direct linear algebra methods such as LU decomposi









      Some useful matrix and vector expressions

      -

      The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and -matrices as upper case boldfaced letters. -

      +

      See the handwritten notes at https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/2022/NotesExercise5Week452022.pdf

      -$$ -\frac{\partial (\boldsymbol{b}^T\boldsymbol{a})}{\partial \boldsymbol{a}} = \boldsymbol{b}, -$$ - -$$ -\frac{\partial (\boldsymbol{a}^T\boldsymbol{A}\boldsymbol{a})}{\partial \boldsymbol{a}} = (\boldsymbol{A}+\boldsymbol{A}^T)\boldsymbol{a}, -$$ - -$$ -\frac{\partial tr(\boldsymbol{B}\boldsymbol{A})}{\partial \boldsymbol{A}} = \boldsymbol{B}^T, -$$ - -$$ -\frac{\partial \log{\vert\boldsymbol{A}\vert}}{\partial \boldsymbol{A}} = (\boldsymbol{A}^{-1})^T. -$$ +

      These notes will be discussed during one of the lectures.











      Interpretations and optimizing our parameters

      @@ -3907,7 +3686,8 @@ ypredict = X_test @ beta









      Exercises

      -

      Here are three possible exercises for weeks 34 and 35.

      + +

      Here are three possible exercises for week 34

      Exercise 1: Setting up various Python environments

      @@ -4042,181 +3822,19 @@ $$ Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits.

      - -

      -Solution. -The code here is an example of where we define our own design matrix and fit parameters \( \beta \). -

      - - -
      -
      -
      -
      -
      -
      import os
      -import numpy as np
      -import pandas as pd
      -import matplotlib.pyplot as plt
      -from sklearn.model_selection import train_test_split
      -
      -def save_fig(fig_id):
      -    plt.savefig(image_path(fig_id) + ".png", format='png')
      -
      -def R2(y_data, y_model):
      -    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
      -def MSE(y_data,y_model):
      -    n = np.size(y_model)
      -    return np.sum((y_data-y_model)**2)/n
      -
      -x = np.random.rand(100)
      -y = 2.0+5*x*x+0.1*np.random.randn(100)
      -
      -
      -#  The design matrix now as function of a given polynomial
      -X = np.zeros((len(x),3))
      -X[:,0] = 1.0
      -X[:,1] = x
      -X[:,2] = x**2
      -# We split the data in test and training data
      -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
      -# matrix inversion to find beta
      -beta = np.linalg.inv(X_train.T @ X_train) @ X_train.T @ y_train
      -print(beta)
      -# and then make the prediction
      -ytilde = X_train @ beta
      -print("Training R2")
      -print(R2(y_train,ytilde))
      -print("Training MSE")
      -print(MSE(y_train,ytilde))
      -ypredict = X_test @ beta
      -print("Test R2")
      -print(R2(y_test,ypredict))
      -print("Test MSE")
      -print(MSE(y_test,ypredict))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - - - - -

      Exercise 3: Normalizing our data

      - -

      A much used approach before starting to train the data is to preprocess our -data. Normally the data may need a rescaling and/or may be sensitive -to extreme values. Scaling the data renders our inputs much more -suitable for the algorithms we want to employ. -

      - -

      Scikit-Learn has several functions which allow us to rescale the -data, normally resulting in much better results in terms of various -accuracy scores. The StandardScaler function in Scikit-Learn -ensures that for each feature/predictor we study the mean value is -zero and the variance is one (every column in the design/feature -matrix). This scaling has the drawback that it does not ensure that -we have a particular maximum or minimum in our data set. Another -function included in Scikit-Learn is the MinMaxScaler which -ensures that all features are exactly between \( 0 \) and \( 1 \). The -

      - -

      The Normalizer scales each data -point such that the feature vector has a euclidean length of one. In other words, it -projects a data point on the circle (or sphere in the case of higher dimensions) with a -radius of 1. This means every data point is scaled by a different number (by the -inverse of it’s length). -This normalization is often used when only the direction (or angle) of the data matters, -not the length of the feature vector. -

      - -

      The RobustScaler works similarly to the StandardScaler in that it -ensures statistical properties for each feature that guarantee that -they are on the same scale. However, the RobustScaler uses the median -and quartiles, instead of mean and variance. This makes the -RobustScaler ignore data points that are very different from the rest -(like measurement errors). These odd data points are also called -outliers, and might often lead to trouble for other scaling -techniques. -

      - -

      It also common to split the data in a training set and a testing set. A typical split is to use \( 80\% \) of the data for training and the rest -for testing. This can be done as follows with our design matrix \( \boldsymbol{X} \) and data \( \boldsymbol{y} \) (remember to import scikit-learn) -

      - - -
      -
      -
      -
      -
      -
      # split in training and test data
      -X_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.2)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Then we can use the standard scaler to scale our data as

      - - -
      -
      -
      -
      -
      -
      scaler = StandardScaler()
      -scaler.fit(X_train)
      -X_train_scaled = scaler.transform(X_train)
      -X_test_scaled = scaler.transform(X_test)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      +

      Exercise 3: Split data in test and training data

      In this exercise we want you to to compute the MSE for the training data and the test data as function of the complexity of a polynomial, -that is the degree of a given polynomial. We want you also to compute the \( R2 \) score as function of the complexity of the model for both training data and test data. You should also run the calculation with and without scaling. +that is the degree of a given polynomial.

      -

      One of -the aims is to reproduce Figure 2.11 of Hastie et al. -

      +

      The aim is to reproduce Figure 2.11 of Hastie et al.

      -

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points.

      +

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points. You should try to vary \( n \) in your analysis.

      @@ -4226,7 +3844,6 @@ the aims is to reproduce Figure 2.11 of
      np.random.seed()
       n = 100
      -maxdegree = 14
       # Make data set.
       x = np.linspace(-3, 3, n).reshape(-1, 1)
       y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
      @@ -4250,7 +3867,7 @@ y = np.exp(-x**2) + SymPy is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS)  and is entirely written in Python. 

      -

      Finally, if you wish to use the light mark-up language -doconce you can convert a standard ascii text file into various HTML -formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using doconce. -

      +

      Finally, we recommend strongly using Autograd or JAX for automatic differentiation.











      Numpy examples and Important Matrix and vector handling packages

      @@ -1088,82 +1074,6 @@ developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly he
    10904. LAPACK:package for solving symmetric, unsymmetric and generalized eigenvalue problems. From LAPACK's website http://www.netlib.org it is possible to download for free all source codes from this library. Both C/C++ and Fortran versions are available.
    10905. BLAS (I, II and III): (Basic Linear Algebra Subprograms) are routines that provide standard building blocks for performing basic vector and matrix operations. Blas I is vector operations, II vector-matrix operations and III matrix-matrix operations. Highly parallelized and efficient codes, all available for download from http://www.netlib.org.
    10906. -









      -

      Basic Matrix Features

      - -
      -Matrix properties reminder -

      -$$ - \mathbf{A} = - \begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\ - a_{21} & a_{22} & a_{23} & a_{24} \\ - a_{31} & a_{32} & a_{33} & a_{34} \\ - a_{41} & a_{42} & a_{43} & a_{44} - \end{bmatrix}\qquad -\mathbf{I} = - \begin{bmatrix} 1 & 0 & 0 & 0 \\ - 0 & 1 & 0 & 0 \\ - 0 & 0 & 1 & 0 \\ - 0 & 0 & 0 & 1 - \end{bmatrix} -$$ - -

      The inverse of a matrix is defined by

      - -$$ -\mathbf{A}^{-1} \cdot \mathbf{A} = I -$$ - - - - - - - - - - - - - -
      Relations Name matrix elements
      \( A=A^{T} \) symmetric \( a_{ij}=a_{ji} \)
      \( A=\left (A^{T}\right )^{-1} \) real orthogonal \( \sum_k a_{ik}a_{jk}=\sum_k a_{ki} a_{kj}=\delta_{ij} \)
      \( A=A^* \) real matrix \( a_{ij}=a_{ij}^* \)
      \( A=A^{\dagger} \) hermitian \( a_{ij}=a_{ji}^* \)
      \( A=\left(A^{\dagger}\right )^{-1} \) unitary \( \sum_k a_{ik}a_{jk}^*=\sum_k a_{ki}^* a_{kj}=\delta_{ij} \)
      -
      - - -









      -

      Some famous Matrices

      - -
        -
      • Diagonal if \( a_{ij}=0 \) for \( i\ne j \)
      • -
      • Upper triangular if \( a_{ij}=0 \) for \( i>j \)
      • -
      • Lower triangular if \( a_{ij}=0 \) for \( i < j \)
      • -
      • Upper Hessenberg if \( a_{ij}=0 \) for \( i>j+1 \)
      • -
      • Lower Hessenberg if \( a_{ij}=0 \) for \( i < j+1 \)
      • -
      • Tridiagonal if \( a_{ij}=0 \) for \( |i -j|>1 \)
      • -
      • Lower banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i>j+p \)
      • -
      • Upper banded with bandwidth \( p \): \( a_{ij}=0 \) for \( i < j+p \)
      • -
      • Banded, block upper triangular, block lower triangular....
      • -
      -









      -

      More Basic Matrix Features

      - -
      -Some Equivalent Statements -

      -

      For an \( N\times N \) matrix \( \mathbf{A} \) the following properties are all equivalent

      - -
        -
      • If the inverse of \( \mathbf{A} \) exists, \( \mathbf{A} \) is nonsingular.
      • -
      • The equation \( \mathbf{Ax}=0 \) implies \( \mathbf{x}=0 \).
      • -
      • The rows of \( \mathbf{A} \) form a basis of \( R^N \).
      • -
      • The columns of \( \mathbf{A} \) form a basis of \( R^N \).
      • -
      • \( \mathbf{A} \) is a product of elementary matrices.
      • -
      • \( 0 \) is not eigenvalue of \( \mathbf{A} \).
      • -
      -
      - -









      Numpy and arrays

      Numpy provides an easy way to handle arrays in Python. The standard way to import this library is as

      @@ -1912,13 +1822,6 @@ As we will see below it leads also to a very concice code close to the mathemati For multidimensional arrays, we recommend strongly xarray. xarray has much of the same flexibility as pandas, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both pandas and xarray.

      -









      -

      Friday August 27

      - -

      "Video of Lecture August 27, 2021":"https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h21/forelesningsvideoer/LectureThursdayAugust27.mp4?vrtx=view-as-webpage

      - -

      Video of Lecture from fall 2020 and Handwritten notes

      -









      Simple linear regression model using scikit-learn

      @@ -2251,62 +2154,8 @@ $$

      Here \( \boldsymbol{a}=\boldsymbol{y} - \boldsymbol{\tilde{y}} \).

      We will discuss in more detail these and other functions in the -various lectures. We conclude this part with another example. Instead -of a linear \( x \)-dependence we study now a cubic polynomial and use the -polynomial regression analysis tools of scikit-learn. +various lectures and lab sessions.

      - - - -
      -
      -
      -
      -
      -
      import matplotlib.pyplot as plt
      -import numpy as np
      -import random
      -from sklearn.linear_model import Ridge
      -from sklearn.preprocessing import PolynomialFeatures
      -from sklearn.pipeline import make_pipeline
      -from sklearn.linear_model import LinearRegression
      -
      -x=np.linspace(0.02,0.98,200)
      -noise = np.asarray(random.sample((range(200)),200))
      -y=x**3*noise
      -yn=x**3*100
      -poly3 = PolynomialFeatures(degree=3)
      -X = poly3.fit_transform(x[:,np.newaxis])
      -clf3 = LinearRegression()
      -clf3.fit(X,y)
      -
      -Xplot=poly3.fit_transform(x[:,np.newaxis])
      -poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')
      -plt.plot(x,yn, color='red', label="True Cubic")
      -plt.scatter(x, y, label='Data', color='orange', s=15)
      -plt.legend()
      -plt.show()
      -
      -def error(a):
      -    for i in y:
      -        err=(y-yn)/yn
      -    return abs(np.sum(err))/len(err)
      -
      -print (error(y))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      To our real data: nuclear binding energies. Brief reminder on masses and binding energies

      Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding @@ -2707,60 +2556,6 @@ plt.show()

      -

      Seeing the wood for the trees

      - -

      As a teaser, let us now see how we can do this with decision trees using scikit-learn. Later we will switch to so-called random forests!

      - - - -
      -
      -
      -
      -
      -
      #Decision Tree Regression
      -from sklearn.tree import DecisionTreeRegressor
      -regr_1=DecisionTreeRegressor(max_depth=5)
      -regr_2=DecisionTreeRegressor(max_depth=7)
      -regr_3=DecisionTreeRegressor(max_depth=9)
      -regr_1.fit(X, Energies)
      -regr_2.fit(X, Energies)
      -regr_3.fit(X, Energies)
      -
      -
      -y_1 = regr_1.predict(X)
      -y_2 = regr_2.predict(X)
      -y_3=regr_3.predict(X)
      -Masses['Eapprox'] = y_3
      -# Plot the results
      -plt.figure()
      -plt.plot(A, Energies, color="blue", label="Data", linewidth=2)
      -plt.plot(A, y_1, color="red", label="max_depth=5", linewidth=2)
      -plt.plot(A, y_2, color="green", label="max_depth=7", linewidth=2)
      -plt.plot(A, y_3, color="m", label="max_depth=9", linewidth=2)
      -
      -plt.xlabel("$A$")
      -plt.ylabel("$E$[MeV]")
      -plt.title("Decision Tree Regression")
      -plt.legend()
      -save_fig("Masses2016Trees")
      -plt.show()
      -print(Masses)
      -print(np.mean( (Energies-y_1)**2))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -

      And what about using neural networks?

      The seaborn package allows us to visualize data in an efficient way. Note that we use scikit-learn's multi-layer perceptron (or feed forward neural network) functionality. @@ -2901,7 +2696,7 @@ the linear regression model where \( \boldsymbol{\beta} = [\beta_0, \ld

      -

      In order to understand the relation among the predictors \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), +

      In order to understand the relation among the predictors (or features or properties) \( p \), the set of data \( n \) and the target (outcome, output etc) \( \boldsymbol{y} \), consider the model we discussed for describing nuclear binding energies.

      @@ -3312,25 +3107,9 @@ allow for the usage of direct linear algebra methods such as LU decomposi









      Some useful matrix and vector expressions

      -

      The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and -matrices as upper case boldfaced letters. -

      +

      See the handwritten notes at https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/2022/NotesExercise5Week452022.pdf

      -$$ -\frac{\partial (\boldsymbol{b}^T\boldsymbol{a})}{\partial \boldsymbol{a}} = \boldsymbol{b}, -$$ - -$$ -\frac{\partial (\boldsymbol{a}^T\boldsymbol{A}\boldsymbol{a})}{\partial \boldsymbol{a}} = (\boldsymbol{A}+\boldsymbol{A}^T)\boldsymbol{a}, -$$ - -$$ -\frac{\partial tr(\boldsymbol{B}\boldsymbol{A})}{\partial \boldsymbol{A}} = \boldsymbol{B}^T, -$$ - -$$ -\frac{\partial \log{\vert\boldsymbol{A}\vert}}{\partial \boldsymbol{A}} = (\boldsymbol{A}^{-1})^T. -$$ +

      These notes will be discussed during one of the lectures.











      Interpretations and optimizing our parameters

      @@ -3984,7 +3763,8 @@ ypredict = X_test Exercises -

      Here are three possible exercises for weeks 34 and 35.

      + +

      Here are three possible exercises for week 34

      Exercise 1: Setting up various Python environments

      @@ -4119,181 +3899,19 @@ $$ Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits.

      - -

      -Solution. -The code here is an example of where we define our own design matrix and fit parameters \( \beta \). -

      - - -
      -
      -
      -
      -
      -
      import os
      -import numpy as np
      -import pandas as pd
      -import matplotlib.pyplot as plt
      -from sklearn.model_selection import train_test_split
      -
      -def save_fig(fig_id):
      -    plt.savefig(image_path(fig_id) + ".png", format='png')
      -
      -def R2(y_data, y_model):
      -    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
      -def MSE(y_data,y_model):
      -    n = np.size(y_model)
      -    return np.sum((y_data-y_model)**2)/n
      -
      -x = np.random.rand(100)
      -y = 2.0+5*x*x+0.1*np.random.randn(100)
      -
      -
      -#  The design matrix now as function of a given polynomial
      -X = np.zeros((len(x),3))
      -X[:,0] = 1.0
      -X[:,1] = x
      -X[:,2] = x**2
      -# We split the data in test and training data
      -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
      -# matrix inversion to find beta
      -beta = np.linalg.inv(X_train.T @ X_train) @ X_train.T @ y_train
      -print(beta)
      -# and then make the prediction
      -ytilde = X_train @ beta
      -print("Training R2")
      -print(R2(y_train,ytilde))
      -print("Training MSE")
      -print(MSE(y_train,ytilde))
      -ypredict = X_test @ beta
      -print("Test R2")
      -print(R2(y_test,ypredict))
      -print("Test MSE")
      -print(MSE(y_test,ypredict))
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - - - - -

      Exercise 3: Normalizing our data

      - -

      A much used approach before starting to train the data is to preprocess our -data. Normally the data may need a rescaling and/or may be sensitive -to extreme values. Scaling the data renders our inputs much more -suitable for the algorithms we want to employ. -

      - -

      Scikit-Learn has several functions which allow us to rescale the -data, normally resulting in much better results in terms of various -accuracy scores. The StandardScaler function in Scikit-Learn -ensures that for each feature/predictor we study the mean value is -zero and the variance is one (every column in the design/feature -matrix). This scaling has the drawback that it does not ensure that -we have a particular maximum or minimum in our data set. Another -function included in Scikit-Learn is the MinMaxScaler which -ensures that all features are exactly between \( 0 \) and \( 1 \). The -

      - -

      The Normalizer scales each data -point such that the feature vector has a euclidean length of one. In other words, it -projects a data point on the circle (or sphere in the case of higher dimensions) with a -radius of 1. This means every data point is scaled by a different number (by the -inverse of it’s length). -This normalization is often used when only the direction (or angle) of the data matters, -not the length of the feature vector. -

      - -

      The RobustScaler works similarly to the StandardScaler in that it -ensures statistical properties for each feature that guarantee that -they are on the same scale. However, the RobustScaler uses the median -and quartiles, instead of mean and variance. This makes the -RobustScaler ignore data points that are very different from the rest -(like measurement errors). These odd data points are also called -outliers, and might often lead to trouble for other scaling -techniques. -

      - -

      It also common to split the data in a training set and a testing set. A typical split is to use \( 80\% \) of the data for training and the rest -for testing. This can be done as follows with our design matrix \( \boldsymbol{X} \) and data \( \boldsymbol{y} \) (remember to import scikit-learn) -

      - - -
      -
      -
      -
      -
      -
      # split in training and test data
      -X_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.2)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      - -

      Then we can use the standard scaler to scale our data as

      - - -
      -
      -
      -
      -
      -
      scaler = StandardScaler()
      -scaler.fit(X_train)
      -X_train_scaled = scaler.transform(X_train)
      -X_test_scaled = scaler.transform(X_test)
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      -
      +

      Exercise 3: Split data in test and training data

      In this exercise we want you to to compute the MSE for the training data and the test data as function of the complexity of a polynomial, -that is the degree of a given polynomial. We want you also to compute the \( R2 \) score as function of the complexity of the model for both training data and test data. You should also run the calculation with and without scaling. +that is the degree of a given polynomial.

      -

      One of -the aims is to reproduce Figure 2.11 of Hastie et al. -

      +

      The aim is to reproduce Figure 2.11 of Hastie et al.

      -

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points.

      +

      Our data is defined by \( x\in [-3,3] \) with a total of for example \( 100 \) data points. You should try to vary \( n \) in your analysis.

      @@ -4303,7 +3921,6 @@ the aims is to reproduce Figure 2.11 of
      np.random.seed()
       n = 100
      -maxdegree = 14
       # Make data set.
       x = np.linspace(-3, 3, n).reshape(-1, 1)
       y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
      @@ -4327,7 +3944,7 @@ y = np.e
       
       

      a) -Write a first code which sets up a design matrix \( X \) defined by a fifth-order polynomial. Scale your data and split it in training and test data. +Write a first code which sets up a design matrix \( X \) defined by a fifth-order polynomial and split your data set in training and test data.

      @@ -4335,7 +3952,7 @@ Write a first code which sets up a design matrix \( X \) defined by a fifth-orde

      b) -Perform an ordinary least squares and compute the means squared error and the \( R2 \) factor for the training data and the test data, with and without scaling. +Perform an ordinary least squares fitting and compute the means squared error for the training data and the test data.

      @@ -4343,7 +3960,7 @@ Perform an ordinary least squares and compute the means squared error and the \(

      c) -Add now a model which allows you to make polynomials up to degree \( 15 \). Perform a standard OLS fitting of the training data and compute the MSE and \( R2 \) for the training and test data and plot both test and training data MSE and \( R2 \) as functions of the polynomial degree. Compare what you see with Figure 2.11 of Hastie et al. Comment your results. For which polynomial degree do you find an optimal MSE (smallest value)? +Add now a model which allows you to make polynomials up to degree \( 15 \). Perform a standard OLS fitting of the training data and compute the MSE for the training and test data and plot both test and training data MSE as functions of the polynomial degree. Compare what you see with Figure 2.11 of Hastie et al. Comment your results. For which polynomial degree do you find an optimal MSE (smallest value)?

      diff --git a/doc/pub/week34/ipynb/ipynb-week34-src.tar.gz b/doc/pub/week34/ipynb/ipynb-week34-src.tar.gz index ad652e490a73db69d3f61dad65f45577eac56fdb..8bfb3b048a509ac72044be6e0f5a14db947d111b 100644 GIT binary patch delta 21 dcmcb!g6+-\n", @@ -12,8 +14,10 @@ }, { "cell_type": "markdown", - "id": "fcd73f8e", - "metadata": {}, + "id": "ad682ef2", + "metadata": { + "editable": true + }, "source": [ "# Week 34: Introduction to the course, Logistics and Practicalities\n", "**Morten Hjorth-Jensen**, Department of Physics and Center for Computing in Science Education, University of Oslo, Norway and Department of Physics and Astronomy and Facility for Rare Isotope Beams, Michigan State University, USA\n", @@ -23,14 +27,16 @@ }, { "cell_type": "markdown", - "id": "e1da964a", - "metadata": {}, + "id": "a3c9b4cb", + "metadata": { + "editable": true + }, "source": [ "## Overview of first week\n", "\n", "1. The sessions on Tuesdays and Wednesdays last four hours for each group (four groups in total) and will include lectures in a flipped mode (promoting active learning) and work on exercices and projects.\n", "\n", - "2. The sessions will begin with lectures, discussions, questions and answers about the material to be covered every week.\n", + "2. The sessions will begin with lectures, discussions, questions and answers about the material to be covered every week. Videos and teaching material will be announced in due time.\n", "\n", "3. There are four groups:\n", "\n", @@ -50,8 +56,10 @@ }, { "cell_type": "markdown", - "id": "cbeb2f85", - "metadata": {}, + "id": "ca50e792", + "metadata": { + "editable": true + }, "source": [ "## Schedule first week\n", "\n", @@ -64,8 +72,10 @@ }, { "cell_type": "markdown", - "id": "4835b20a", - "metadata": {}, + "id": "63fcf456", + "metadata": { + "editable": true + }, "source": [ "## Reading Recommendations\n", "\n", @@ -85,8 +95,10 @@ }, { "cell_type": "markdown", - "id": "3e30978b", - "metadata": {}, + "id": "4d7d32a1", + "metadata": { + "editable": true + }, "source": [ "## Lectures and ComputerLab\n", "\n", @@ -107,8 +119,10 @@ }, { "cell_type": "markdown", - "id": "7917611c", - "metadata": {}, + "id": "2421a034", + "metadata": { + "editable": true + }, "source": [ "## Communication channels\n", "\n", @@ -119,8 +133,10 @@ }, { "cell_type": "markdown", - "id": "7ccfe004", - "metadata": {}, + "id": "0675b78c", + "metadata": { + "editable": true + }, "source": [ "## Course Format\n", "\n", @@ -141,8 +157,10 @@ }, { "cell_type": "markdown", - "id": "1a747992", - "metadata": {}, + "id": "53e77356", + "metadata": { + "editable": true + }, "source": [ "## Teachers\n", "\n", @@ -170,8 +188,10 @@ }, { "cell_type": "markdown", - "id": "4410a68b", - "metadata": {}, + "id": "0aa0c646", + "metadata": { + "editable": true + }, "source": [ "## Deadlines for projects (tentative)\n", "\n", @@ -186,8 +206,10 @@ }, { "cell_type": "markdown", - "id": "19812ab6", - "metadata": {}, + "id": "c03f73ef", + "metadata": { + "editable": true + }, "source": [ "## Recommended textbooks\n", "\n", @@ -208,8 +230,10 @@ }, { "cell_type": "markdown", - "id": "66147219", - "metadata": {}, + "id": "83371a9e", + "metadata": { + "editable": true + }, "source": [ "## Prerequisites\n", "\n", @@ -226,8 +250,10 @@ }, { "cell_type": "markdown", - "id": "42a134be", - "metadata": {}, + "id": "1d9b6bd4", + "metadata": { + "editable": true + }, "source": [ "## Learning outcomes\n", "\n", @@ -268,8 +294,10 @@ }, { "cell_type": "markdown", - "id": "aa87b03b", - "metadata": {}, + "id": "d2b3cdf9", + "metadata": { + "editable": true + }, "source": [ "## Topics covered in this course: Statistical analysis and optimization of data\n", "\n", @@ -301,8 +329,10 @@ }, { "cell_type": "markdown", - "id": "8cca3e5b", - "metadata": {}, + "id": "26eea71c", + "metadata": { + "editable": true + }, "source": [ "## Topics covered in this course\n", "\n", @@ -337,8 +367,10 @@ }, { "cell_type": "markdown", - "id": "6f1b8ac6", - "metadata": {}, + "id": "850df98b", + "metadata": { + "editable": true + }, "source": [ "## Extremely useful tools, strongly recommended\n", "\n", @@ -351,8 +383,10 @@ }, { "cell_type": "markdown", - "id": "2d590bee", - "metadata": {}, + "id": "ccaadbac", + "metadata": { + "editable": true + }, "source": [ "## Other courses on Data science and Machine Learning at UiO\n", "\n", @@ -380,8 +414,10 @@ }, { "cell_type": "markdown", - "id": "687b1a20", - "metadata": {}, + "id": "fbb62679", + "metadata": { + "editable": true + }, "source": [ "## Introduction\n", "\n", @@ -411,9 +447,9 @@ "topics and tools as well as showing the power of various Python \n", "libraries for machine learning and statistical data analysis. \n", "\n", - "Here, we will mainly focus on two\n", + "Although we have projects where you write your own codes, we will also focus on two\n", "specific Python packages for Machine Learning, Scikit-Learn and\n", - "Tensorflow (see below for links etc). Moreover, the examples we\n", + "Tensorflow with Keras (see below for links etc). Moreover, the examples we\n", "introduce will serve as inputs to many of our discussions later, as\n", "well as allowing you to set up models and produce your own data and\n", "get started with programming." @@ -421,8 +457,10 @@ }, { "cell_type": "markdown", - "id": "09496a03", - "metadata": {}, + "id": "1b51b4ee", + "metadata": { + "editable": true + }, "source": [ "## What is Machine Learning?\n", "\n", @@ -492,8 +530,10 @@ }, { "cell_type": "markdown", - "id": "07f9251d", - "metadata": {}, + "id": "777bf300", + "metadata": { + "editable": true + }, "source": [ "## Types of Machine Learning\n", "\n", @@ -519,8 +559,10 @@ }, { "cell_type": "markdown", - "id": "b6ab5119", - "metadata": {}, + "id": "d4f5891d", + "metadata": { + "editable": true + }, "source": [ "## Essential elements of ML\n", "\n", @@ -535,8 +577,10 @@ }, { "cell_type": "markdown", - "id": "4e705506", - "metadata": {}, + "id": "5d8b4a2f", + "metadata": { + "editable": true + }, "source": [ "## An optimization/minimization problem\n", "\n", @@ -545,8 +589,10 @@ }, { "cell_type": "markdown", - "id": "114abc74", - "metadata": {}, + "id": "4fc2547d", + "metadata": { + "editable": true + }, "source": [ "## A Frequentist approach to data analysis\n", "\n", @@ -578,8 +624,10 @@ }, { "cell_type": "markdown", - "id": "8b2bbf6d", - "metadata": {}, + "id": "b134aa47", + "metadata": { + "editable": true + }, "source": [ "## What is a good model?\n", "\n", @@ -606,8 +654,10 @@ }, { "cell_type": "markdown", - "id": "4e64bf89", - "metadata": {}, + "id": "54c53efd", + "metadata": { + "editable": true + }, "source": [ "## What is a good model? Can we define it?\n", "\n", @@ -635,8 +685,10 @@ }, { "cell_type": "markdown", - "id": "5483611c", - "metadata": {}, + "id": "b25ea1d1", + "metadata": { + "editable": true + }, "source": [ "## Software and needed installations\n", "\n", @@ -672,8 +724,10 @@ }, { "cell_type": "markdown", - "id": "9eac6e63", - "metadata": {}, + "id": "37eba906", + "metadata": { + "editable": true + }, "source": [ "## Python installers\n", "\n", @@ -703,8 +757,10 @@ }, { "cell_type": "markdown", - "id": "4b4f5215", - "metadata": {}, + "id": "1fd5ce9b", + "metadata": { + "editable": true + }, "source": [ "## Useful Python libraries\n", "Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there)\n", @@ -736,8 +792,10 @@ }, { "cell_type": "markdown", - "id": "95ed9491", - "metadata": {}, + "id": "7a563428", + "metadata": { + "editable": true + }, "source": [ "## Installing R, C++, cython or Julia\n", "\n", @@ -746,8 +804,8 @@ "Those of you\n", "already familiar with **R** should feel free to continue using **R**, keeping\n", "however an eye on the parallel Python set ups. Similarly, if you are a\n", - "Python afecionado, feel free to explore **R** as well. Jupyter/Ipython\n", - "notebook allows you to run **R** codes interactively in your\n", + "Python afecionado, feel free to explore **R** as well. Jupyter(Julia, Python and R) /Ipython\n", + "notebook allows you to run **R** codes and **Julia** codes interactively in your\n", "browser. The software library **R** is really tailored for statistical data analysis\n", "and allows for an easy usage of the tools and algorithms we will discuss in these\n", "lectures.\n", @@ -758,8 +816,10 @@ }, { "cell_type": "markdown", - "id": "374079cd", - "metadata": {}, + "id": "67c90d71", + "metadata": { + "editable": true + }, "source": [ "## Installing R, C++, cython, Numba etc\n", "\n", @@ -783,28 +843,32 @@ }, { "cell_type": "markdown", - "id": "af22d31e", - "metadata": {}, + "id": "23ec7d03", + "metadata": { + "editable": true + }, "source": [ " pycod jupyter nbconvert filename.ipynb --to latex \n" ] }, { "cell_type": "markdown", - "id": "6a65b86c", - "metadata": {}, + "id": "7bf3cec8", + "metadata": { + "editable": true + }, "source": [ "And to add more versatility, the Python package [SymPy](http://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python. \n", "\n", - "Finally, if you wish to use the light mark-up language \n", - "[doconce](https://github.com/hplgit/doconce) you can convert a standard ascii text file into various HTML \n", - "formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using **doconce**." + "Finally, we recommend strongly using Autograd or JAX for automatic differentiation." ] }, { "cell_type": "markdown", - "id": "ddd0d581", - "metadata": {}, + "id": "2ffa9ff0", + "metadata": { + "editable": true + }, "source": [ "## Numpy examples and Important Matrix and vector handling packages\n", "\n", @@ -822,126 +886,10 @@ }, { "cell_type": "markdown", - "id": "9fa618fa", - "metadata": {}, - "source": [ - "## Basic Matrix Features\n", - "\n", - "**Matrix properties reminder.**" - ] - }, - { - "cell_type": "markdown", - "id": "d849e19e", - "metadata": {}, - "source": [ - "$$\n", - "\\mathbf{A} =\n", - " \\begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\\\\n", - " a_{21} & a_{22} & a_{23} & a_{24} \\\\\n", - " a_{31} & a_{32} & a_{33} & a_{34} \\\\\n", - " a_{41} & a_{42} & a_{43} & a_{44}\n", - " \\end{bmatrix}\\qquad\n", - "\\mathbf{I} =\n", - " \\begin{bmatrix} 1 & 0 & 0 & 0 \\\\\n", - " 0 & 1 & 0 & 0 \\\\\n", - " 0 & 0 & 1 & 0 \\\\\n", - " 0 & 0 & 0 & 1\n", - " \\end{bmatrix}\n", - "$$" - ] - }, - { - "cell_type": "markdown", - "id": "43dd13c9", - "metadata": {}, - "source": [ - "The inverse of a matrix is defined by" - ] - }, - { - "cell_type": "markdown", - "id": "d5b42841", - "metadata": {}, - "source": [ - "$$\n", - "\\mathbf{A}^{-1} \\cdot \\mathbf{A} = I\n", - "$$" - ] - }, - { - "cell_type": "markdown", - "id": "d2756d9d", - "metadata": {}, - "source": [ - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "
      Relations Name matrix elements
      $A=A^{T}$ symmetric $a_{ij}=a_{ji}$
      $A=\\left (A^{T}\\right )^{-1}$ real orthogonal $\\sum_k a_{ik}a_{jk}=\\sum_k a_{ki} a_{kj}=\\delta_{ij}$
      $A=A^*$ real matrix $a_{ij}=a_{ij}^*$
      $A=A^{\\dagger}$ hermitian $a_{ij}=a_{ji}^*$
      $A=\\left(A^{\\dagger}\\right )^{-1}$ unitary $\\sum_k a_{ik}a_{jk}^*=\\sum_k a_{ki}^* a_{kj}=\\delta_{ij}$
      " - ] - }, - { - "cell_type": "markdown", - "id": "f7f5934c", - "metadata": {}, - "source": [ - "### Some famous Matrices\n", - "\n", - " * Diagonal if $a_{ij}=0$ for $i\\ne j$\n", - "\n", - " * Upper triangular if $a_{ij}=0$ for $i>j$\n", - "\n", - " * Lower triangular if $a_{ij}=0$ for $ij+1$\n", - "\n", - " * Lower Hessenberg if $a_{ij}=0$ for $i1$\n", - "\n", - " * Lower banded with bandwidth $p$: $a_{ij}=0$ for $i>j+p$\n", - "\n", - " * Upper banded with bandwidth $p$: $a_{ij}=0$ for $i\n", + "\n", + "These notes will be discussed during one of the lectures." ] }, { "cell_type": "markdown", - "id": "888346fc", - "metadata": {}, - "source": [ - "$$\n", - "\\frac{\\partial (\\boldsymbol{b}^T\\boldsymbol{a})}{\\partial \\boldsymbol{a}} = \\boldsymbol{b},\n", - "$$" - ] - }, - { - "cell_type": "markdown", - "id": "0b88b7f5", - "metadata": {}, - "source": [ - "$$\n", - "\\frac{\\partial (\\boldsymbol{a}^T\\boldsymbol{A}\\boldsymbol{a})}{\\partial \\boldsymbol{a}} = (\\boldsymbol{A}+\\boldsymbol{A}^T)\\boldsymbol{a},\n", - "$$" - ] - }, - { - "cell_type": "markdown", - "id": "0108eefb", - "metadata": {}, - "source": [ - "$$\n", - "\\frac{\\partial tr(\\boldsymbol{B}\\boldsymbol{A})}{\\partial \\boldsymbol{A}} = \\boldsymbol{B}^T,\n", - "$$" - ] - }, - { - "cell_type": "markdown", - "id": "1c3aeec5", - "metadata": {}, - "source": [ - "$$\n", - "\\frac{\\partial \\log{\\vert\\boldsymbol{A}\\vert}}{\\partial \\boldsymbol{A}} = (\\boldsymbol{A}^{-1})^T.\n", - "$$" - ] - }, - { - "cell_type": "markdown", - "id": "0c808963", - "metadata": {}, + "id": "1158b8d6", + "metadata": { + "editable": true + }, "source": [ "## Interpretations and optimizing our parameters\n", "The residuals $\\boldsymbol{\\epsilon}$ are in turn given by" @@ -3482,8 +3663,10 @@ }, { "cell_type": "markdown", - "id": "58a7a52b", - "metadata": {}, + "id": "633e1d2c", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\boldsymbol{\\epsilon} = \\boldsymbol{y}-\\boldsymbol{\\tilde{y}} = \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta},\n", @@ -3492,16 +3675,20 @@ }, { "cell_type": "markdown", - "id": "31380692", - "metadata": {}, + "id": "f81d31f8", + "metadata": { + "editable": true + }, "source": [ "and with" ] }, { "cell_type": "markdown", - "id": "338d57bf", - "metadata": {}, + "id": "dc428bfa", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)= 0,\n", @@ -3510,16 +3697,20 @@ }, { "cell_type": "markdown", - "id": "83acfd23", - "metadata": {}, + "id": "8713d26e", + "metadata": { + "editable": true + }, "source": [ "we have" ] }, { "cell_type": "markdown", - "id": "e1096ec3", - "metadata": {}, + "id": "6da49a7a", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\boldsymbol{X}^T\\boldsymbol{\\epsilon}=\\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)= 0,\n", @@ -3528,8 +3719,10 @@ }, { "cell_type": "markdown", - "id": "cd841096", - "metadata": {}, + "id": "82436bba", + "metadata": { + "editable": true + }, "source": [ "meaning that the solution for $\\boldsymbol{\\beta}$ is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach.\n", "\n", @@ -3538,8 +3731,10 @@ }, { "cell_type": "markdown", - "id": "a5eb6867", - "metadata": {}, + "id": "3221853e", + "metadata": { + "editable": true + }, "source": [ "## Own code for Ordinary Least Squares\n", "\n", @@ -3549,9 +3744,12 @@ }, { "cell_type": "code", - "execution_count": 39, - "id": "0c33b4a6", - "metadata": {}, + "execution_count": 37, + "id": "32634369", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "# matrix inversion to find beta\n", @@ -3562,17 +3760,22 @@ }, { "cell_type": "markdown", - "id": "30b8c643", - "metadata": {}, + "id": "5fdb2784", + "metadata": { + "editable": true + }, "source": [ "Alternatively, you can use the least squares functionality in **Numpy** as" ] }, { "cell_type": "code", - "execution_count": 40, - "id": "2f1c46d5", - "metadata": {}, + "execution_count": 38, + "id": "0a7052de", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "fit = np.linalg.lstsq(X, Energies, rcond =None)[0]\n", @@ -3581,17 +3784,22 @@ }, { "cell_type": "markdown", - "id": "902a28cd", - "metadata": {}, + "id": "0c3fadf2", + "metadata": { + "editable": true + }, "source": [ "And finally we plot our fit with and compare with data" ] }, { "cell_type": "code", - "execution_count": 41, - "id": "5c5d6920", - "metadata": {}, + "execution_count": 39, + "id": "49865a8e", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "Masses['Eapprox'] = ytilde\n", @@ -3610,8 +3818,10 @@ }, { "cell_type": "markdown", - "id": "c6ab2ba2", - "metadata": {}, + "id": "d0d445d7", + "metadata": { + "editable": true + }, "source": [ "## Adding error analysis and training set up\n", "\n", @@ -3621,9 +3831,12 @@ }, { "cell_type": "code", - "execution_count": 42, - "id": "cb5a5325", - "metadata": {}, + "execution_count": 40, + "id": "d15cdce6", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "def R2(y_data, y_model):\n", @@ -3632,17 +3845,22 @@ }, { "cell_type": "markdown", - "id": "25b4cfd8", - "metadata": {}, + "id": "2275cd6d", + "metadata": { + "editable": true + }, "source": [ "and we would be using it as" ] }, { "cell_type": "code", - "execution_count": 43, - "id": "5c2cc37e", - "metadata": {}, + "execution_count": 41, + "id": "f07b786c", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "print(R2(Energies,ytilde))" @@ -3650,17 +3868,22 @@ }, { "cell_type": "markdown", - "id": "463a5289", - "metadata": {}, + "id": "0af606bc", + "metadata": { + "editable": true + }, "source": [ "We can easily add our **MSE** score as" ] }, { "cell_type": "code", - "execution_count": 44, - "id": "a5f77d4b", - "metadata": {}, + "execution_count": 42, + "id": "0df8ce2d", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "def MSE(y_data,y_model):\n", @@ -3672,17 +3895,22 @@ }, { "cell_type": "markdown", - "id": "4ac24929", - "metadata": {}, + "id": "70e31542", + "metadata": { + "editable": true + }, "source": [ "and finally the relative error as" ] }, { "cell_type": "code", - "execution_count": 45, - "id": "231ad359", - "metadata": {}, + "execution_count": 43, + "id": "b22073d9", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "def RelativeError(y_data,y_model):\n", @@ -3692,8 +3920,10 @@ }, { "cell_type": "markdown", - "id": "d826bba2", - "metadata": {}, + "id": "36b83437", + "metadata": { + "editable": true + }, "source": [ "## The $\\chi^2$ function\n", "\n", @@ -3712,8 +3942,10 @@ }, { "cell_type": "markdown", - "id": "0a66d959", - "metadata": {}, + "id": "a29b9c4c", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\chi^2(\\boldsymbol{\\beta})=\\frac{1}{n}\\sum_{i=0}^{n-1}\\frac{\\left(y_i-\\tilde{y}_i\\right)^2}{\\sigma_i^2}=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)^T\\frac{1}{\\boldsymbol{\\Sigma^2}}\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)\\right\\},\n", @@ -3722,16 +3954,20 @@ }, { "cell_type": "markdown", - "id": "f9be426f", - "metadata": {}, + "id": "488d3565", + "metadata": { + "editable": true + }, "source": [ "where the matrix $\\boldsymbol{\\Sigma}$ is a diagonal matrix with $\\sigma_i$ as matrix elements." ] }, { "cell_type": "markdown", - "id": "bc82c650", - "metadata": {}, + "id": "7fdcf0e7", + "metadata": { + "editable": true + }, "source": [ "## The $\\chi^2$ function\n", "\n", @@ -3740,8 +3976,10 @@ }, { "cell_type": "markdown", - "id": "28087a6d", - "metadata": {}, + "id": "412c2315", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_j} = \\frac{\\partial }{\\partial \\beta_j}\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(\\frac{y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}}{\\sigma_i}\\right)^2\\right]=0,\n", @@ -3750,16 +3988,20 @@ }, { "cell_type": "markdown", - "id": "edb55e07", - "metadata": {}, + "id": "4fc8aa35", + "metadata": { + "editable": true + }, "source": [ "which results in" ] }, { "cell_type": "markdown", - "id": "aaded591", - "metadata": {}, + "id": "80b19a0f", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_j} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}\\frac{x_{ij}}{\\sigma_i}\\left(\\frac{y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}}{\\sigma_i}\\right)\\right]=0,\n", @@ -3768,16 +4010,20 @@ }, { "cell_type": "markdown", - "id": "5308f0fd", - "metadata": {}, + "id": "58e46ba8", + "metadata": { + "editable": true + }, "source": [ "or in a matrix-vector form as" ] }, { "cell_type": "markdown", - "id": "cdbeb539", - "metadata": {}, + "id": "c5f6428e", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{A}^T\\left( \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{\\beta}\\right).\n", @@ -3786,16 +4032,20 @@ }, { "cell_type": "markdown", - "id": "6c3176b9", - "metadata": {}, + "id": "026e3e70", + "metadata": { + "editable": true + }, "source": [ "where we have defined the matrix $\\boldsymbol{A} =\\boldsymbol{X}/\\boldsymbol{\\Sigma}$ with matrix elements $a_{ij} = x_{ij}/\\sigma_i$ and the vector $\\boldsymbol{b}$ with elements $b_i = y_i/\\sigma_i$." ] }, { "cell_type": "markdown", - "id": "c9171b84", - "metadata": {}, + "id": "051702cb", + "metadata": { + "editable": true + }, "source": [ "## The $\\chi^2$ function\n", "\n", @@ -3804,8 +4054,10 @@ }, { "cell_type": "markdown", - "id": "005b23f5", - "metadata": {}, + "id": "1e532239", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{A}^T\\left( \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{\\beta}\\right),\n", @@ -3814,16 +4066,20 @@ }, { "cell_type": "markdown", - "id": "011b1cf3", - "metadata": {}, + "id": "5f5358c1", + "metadata": { + "editable": true + }, "source": [ "as" ] }, { "cell_type": "markdown", - "id": "eecb33b0", - "metadata": {}, + "id": "8aa7f812", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\boldsymbol{A}^T\\boldsymbol{b} = \\boldsymbol{A}^T\\boldsymbol{A}\\boldsymbol{\\beta},\n", @@ -3832,16 +4088,20 @@ }, { "cell_type": "markdown", - "id": "477dda35", - "metadata": {}, + "id": "3acb7998", + "metadata": { + "editable": true + }, "source": [ "and if the matrix $\\boldsymbol{A}^T\\boldsymbol{A}$ is invertible we have the solution" ] }, { "cell_type": "markdown", - "id": "954eea3f", - "metadata": {}, + "id": "4dd73372", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\boldsymbol{\\beta} =\\left(\\boldsymbol{A}^T\\boldsymbol{A}\\right)^{-1}\\boldsymbol{A}^T\\boldsymbol{b}.\n", @@ -3850,8 +4110,10 @@ }, { "cell_type": "markdown", - "id": "a3513171", - "metadata": {}, + "id": "0a2d1ab2", + "metadata": { + "editable": true + }, "source": [ "## The $\\chi^2$ function\n", "\n", @@ -3860,8 +4122,10 @@ }, { "cell_type": "markdown", - "id": "b61882ef", - "metadata": {}, + "id": "5e740c4d", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\boldsymbol{H} = \\left(\\boldsymbol{A}^T\\boldsymbol{A}\\right)^{-1},\n", @@ -3870,16 +4134,20 @@ }, { "cell_type": "markdown", - "id": "d3b19486", - "metadata": {}, + "id": "f332b423", + "metadata": { + "editable": true + }, "source": [ "we have then the following expression for the parameters $\\beta_j$ (the matrix elements of $\\boldsymbol{H}$ are $h_{ij}$)" ] }, { "cell_type": "markdown", - "id": "f26a6317", - "metadata": {}, + "id": "cab61b59", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\beta_j = \\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}\\frac{y_i}{\\sigma_i}\\frac{x_{ik}}{\\sigma_i} = \\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}b_ia_{ik}\n", @@ -3888,16 +4156,20 @@ }, { "cell_type": "markdown", - "id": "ed55e7b5", - "metadata": {}, + "id": "20b9d29f", + "metadata": { + "editable": true + }, "source": [ "We state without proof the expression for the uncertainty in the parameters $\\beta_j$ as (we leave this as an exercise)" ] }, { "cell_type": "markdown", - "id": "0d4ff691", - "metadata": {}, + "id": "7bfbf5e2", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\sigma^2(\\beta_j) = \\sum_{i=0}^{n-1}\\sigma_i^2\\left( \\frac{\\partial \\beta_j}{\\partial y_i}\\right)^2,\n", @@ -3906,16 +4178,20 @@ }, { "cell_type": "markdown", - "id": "5da9cdc6", - "metadata": {}, + "id": "71366bb6", + "metadata": { + "editable": true + }, "source": [ "resulting in" ] }, { "cell_type": "markdown", - "id": "85883ee8", - "metadata": {}, + "id": "9f0e6e6e", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\sigma^2(\\beta_j) = \\left(\\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}a_{ik}\\right)\\left(\\sum_{l=0}^{p-1}h_{jl}\\sum_{m=0}^{n-1}a_{ml}\\right) = h_{jj}!\n", @@ -3924,8 +4200,10 @@ }, { "cell_type": "markdown", - "id": "c3ac5064", - "metadata": {}, + "id": "c6ecc5fc", + "metadata": { + "editable": true + }, "source": [ "## The $\\chi^2$ function\n", "The first step here is to approximate the function $y$ with a first-order polynomial, that is we write" @@ -3933,8 +4211,10 @@ }, { "cell_type": "markdown", - "id": "3121513b", - "metadata": {}, + "id": "0eaa78b9", + "metadata": { + "editable": true + }, "source": [ "$$\n", "y=y(x) \\rightarrow y(x_i) \\approx \\beta_0+\\beta_1 x_i.\n", @@ -3943,16 +4223,20 @@ }, { "cell_type": "markdown", - "id": "c9fc0e9d", - "metadata": {}, + "id": "89650d96", + "metadata": { + "editable": true + }, "source": [ "By computing the derivatives of $\\chi^2$ with respect to $\\beta_0$ and $\\beta_1$ show that these are given by" ] }, { "cell_type": "markdown", - "id": "50ee610e", - "metadata": {}, + "id": "86b81e9f", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_0} = -2\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(\\frac{y_i-\\beta_0-\\beta_1x_{i}}{\\sigma_i^2}\\right)\\right]=0,\n", @@ -3961,16 +4245,20 @@ }, { "cell_type": "markdown", - "id": "94fba98c", - "metadata": {}, + "id": "c3cdd79a", + "metadata": { + "editable": true + }, "source": [ "and" ] }, { "cell_type": "markdown", - "id": "496da037", - "metadata": {}, + "id": "58bae61c", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_1} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}x_i\\left(\\frac{y_i-\\beta_0-\\beta_1x_{i}}{\\sigma_i^2}\\right)\\right]=0.\n", @@ -3979,8 +4267,10 @@ }, { "cell_type": "markdown", - "id": "9c8f2972", - "metadata": {}, + "id": "2a7c0936", + "metadata": { + "editable": true + }, "source": [ "## The $\\chi^2$ function\n", "\n", @@ -3990,8 +4280,10 @@ }, { "cell_type": "markdown", - "id": "e49831a8", - "metadata": {}, + "id": "b72360a2", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\gamma = \\sum_{i=0}^{n-1}\\frac{1}{\\sigma_i^2},\n", @@ -4000,8 +4292,10 @@ }, { "cell_type": "markdown", - "id": "fb152733", - "metadata": {}, + "id": "89188b6f", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\gamma_x = \\sum_{i=0}^{n-1}\\frac{x_{i}}{\\sigma_i^2},\n", @@ -4010,8 +4304,10 @@ }, { "cell_type": "markdown", - "id": "6bee7350", - "metadata": {}, + "id": "b2483bb6", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\gamma_y = \\sum_{i=0}^{n-1}\\left(\\frac{y_i}{\\sigma_i^2}\\right),\n", @@ -4020,8 +4316,10 @@ }, { "cell_type": "markdown", - "id": "b20214ae", - "metadata": {}, + "id": "19b6390a", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\gamma_{xx} = \\sum_{i=0}^{n-1}\\frac{x_ix_{i}}{\\sigma_i^2},\n", @@ -4030,8 +4328,10 @@ }, { "cell_type": "markdown", - "id": "5b69a59e", - "metadata": {}, + "id": "54ce44a6", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\gamma_{xy} = \\sum_{i=0}^{n-1}\\frac{y_ix_{i}}{\\sigma_i^2},\n", @@ -4040,16 +4340,20 @@ }, { "cell_type": "markdown", - "id": "f3431f6b", - "metadata": {}, + "id": "eeb4a6a2", + "metadata": { + "editable": true + }, "source": [ "we obtain" ] }, { "cell_type": "markdown", - "id": "b4be888f", - "metadata": {}, + "id": "6b8c03ae", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\beta_0 = \\frac{\\gamma_{xx}\\gamma_y-\\gamma_x\\gamma_y}{\\gamma\\gamma_{xx}-\\gamma_x^2},\n", @@ -4058,8 +4362,10 @@ }, { "cell_type": "markdown", - "id": "50edcbf3", - "metadata": {}, + "id": "625f295e", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\beta_1 = \\frac{\\gamma_{xy}\\gamma-\\gamma_x\\gamma_y}{\\gamma\\gamma_{xx}-\\gamma_x^2}.\n", @@ -4068,8 +4374,10 @@ }, { "cell_type": "markdown", - "id": "801af8e7", - "metadata": {}, + "id": "0f0c6fca", + "metadata": { + "editable": true + }, "source": [ "This approach (different linear and non-linear regression) suffers\n", "often from both being underdetermined and overdetermined in the\n", @@ -4079,8 +4387,10 @@ }, { "cell_type": "markdown", - "id": "b6270d66", - "metadata": {}, + "id": "df57fba0", + "metadata": { + "editable": true + }, "source": [ "## Fitting an Equation of State for Dense Nuclear Matter\n", "\n", @@ -4104,17 +4414,22 @@ }, { "cell_type": "markdown", - "id": "3b1011c6", - "metadata": {}, + "id": "14d5df9a", + "metadata": { + "editable": true + }, "source": [ "## The code" ] }, { "cell_type": "code", - "execution_count": 46, - "id": "21e0cf86", - "metadata": {}, + "execution_count": 44, + "id": "90958307", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "# Common imports\n", @@ -4206,8 +4521,10 @@ }, { "cell_type": "markdown", - "id": "b46604cc", - "metadata": {}, + "id": "f42c2ece", + "metadata": { + "editable": true + }, "source": [ "The above simple polynomial in density $\\rho$ gives an excellent fit\n", "to the data. \n", @@ -4219,8 +4536,10 @@ }, { "cell_type": "markdown", - "id": "61d3190d", - "metadata": {}, + "id": "305220dd", + "metadata": { + "editable": true + }, "source": [ "## Splitting our Data in Training and Test data\n", "\n", @@ -4238,9 +4557,12 @@ }, { "cell_type": "code", - "execution_count": 47, - "id": "74ee63ce", - "metadata": {}, + "execution_count": 45, + "id": "70133882", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "import os\n", @@ -4311,17 +4633,22 @@ }, { "cell_type": "markdown", - "id": "bbed49d1", - "metadata": {}, + "id": "b300dcd4", + "metadata": { + "editable": true + }, "source": [ "## Exercises\n", - "Here are three possible exercises for weeks 34 and 35." + "\n", + "Here are three possible exercises for week 34" ] }, { "cell_type": "markdown", - "id": "98b47c09", - "metadata": {}, + "id": "abfe4134", + "metadata": { + "editable": true + }, "source": [ "## Exercise 1: Setting up various Python environments\n", "\n", @@ -4387,8 +4714,10 @@ }, { "cell_type": "markdown", - "id": "53417c28", - "metadata": {}, + "id": "685c34f3", + "metadata": { + "editable": true + }, "source": [ "## Exercise 2: making your own data and exploring scikit-learn\n", "\n", @@ -4398,9 +4727,12 @@ }, { "cell_type": "code", - "execution_count": 48, - "id": "583b5833", - "metadata": {}, + "execution_count": 46, + "id": "4068e855", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "x = np.random.rand(100,1)\n", @@ -4409,8 +4741,10 @@ }, { "cell_type": "markdown", - "id": "c9e4d6a0", - "metadata": {}, + "id": "902ac5a1", + "metadata": { + "editable": true + }, "source": [ "1. Write your own code (following the examples under the [regression notes](https://compphysics.github.io/MachineLearning/doc/LectureNotes/_build/html/chapter1.html)) for computing the parametrization of the data set fitting a second-order polynomial. \n", "\n", @@ -4421,8 +4755,10 @@ }, { "cell_type": "markdown", - "id": "d7763507", - "metadata": {}, + "id": "72d229b7", + "metadata": { + "editable": true + }, "source": [ "$$\n", "MSE(\\boldsymbol{y},\\boldsymbol{\\tilde{y}}) = \\frac{1}{n}\n", @@ -4432,8 +4768,10 @@ }, { "cell_type": "markdown", - "id": "644c77b3", - "metadata": {}, + "id": "60320f76", + "metadata": { + "editable": true + }, "source": [ "and the $R^2$ score function.\n", "If $\\tilde{\\boldsymbol{y}}_i$ is the predicted value of the $i-th$ sample and $y_i$ is the corresponding true value, then the score $R^2$ is defined as" @@ -4441,8 +4779,10 @@ }, { "cell_type": "markdown", - "id": "b0e315d8", - "metadata": {}, + "id": "8ba3e83a", + "metadata": { + "editable": true + }, "source": [ "$$\n", "R^2(\\boldsymbol{y}, \\tilde{\\boldsymbol{y}}) = 1 - \\frac{\\sum_{i=0}^{n - 1} (y_i - \\tilde{y}_i)^2}{\\sum_{i=0}^{n - 1} (y_i - \\bar{y})^2},\n", @@ -4451,16 +4791,20 @@ }, { "cell_type": "markdown", - "id": "8a52d1ec", - "metadata": {}, + "id": "aca39d18", + "metadata": { + "editable": true + }, "source": [ "where we have defined the mean value of $\\boldsymbol{y}$ as" ] }, { "cell_type": "markdown", - "id": "387deb50", - "metadata": {}, + "id": "ae5d4ec6", + "metadata": { + "editable": true + }, "source": [ "$$\n", "\\bar{y} = \\frac{1}{n} \\sum_{i=0}^{n - 1} y_i.\n", @@ -4469,174 +4813,45 @@ }, { "cell_type": "markdown", - "id": "256391d7", - "metadata": {}, + "id": "1a059b8b", + "metadata": { + "editable": true + }, "source": [ "You can use the functionality included in scikit-learn. If you feel for it, you can use your own program and define functions which compute the above two functions. \n", - "Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits.\n", - "\n", - "\n", - "**Solution.**\n", - "The code here is an example of where we define our own design matrix and fit parameters $\\beta$." - ] - }, - { - "cell_type": "code", - "execution_count": 49, - "id": "d5c01490", - "metadata": {}, - "outputs": [], - "source": [ - "import os\n", - "import numpy as np\n", - "import pandas as pd\n", - "import matplotlib.pyplot as plt\n", - "from sklearn.model_selection import train_test_split\n", - "\n", - "def save_fig(fig_id):\n", - " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", - "\n", - "def R2(y_data, y_model):\n", - " return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n", - "def MSE(y_data,y_model):\n", - " n = np.size(y_model)\n", - " return np.sum((y_data-y_model)**2)/n\n", - "\n", - "x = np.random.rand(100)\n", - "y = 2.0+5*x*x+0.1*np.random.randn(100)\n", - "\n", - "\n", - "# The design matrix now as function of a given polynomial\n", - "X = np.zeros((len(x),3))\n", - "X[:,0] = 1.0\n", - "X[:,1] = x\n", - "X[:,2] = x**2\n", - "# We split the data in test and training data\n", - "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n", - "# matrix inversion to find beta\n", - "beta = np.linalg.inv(X_train.T @ X_train) @ X_train.T @ y_train\n", - "print(beta)\n", - "# and then make the prediction\n", - "ytilde = X_train @ beta\n", - "print(\"Training R2\")\n", - "print(R2(y_train,ytilde))\n", - "print(\"Training MSE\")\n", - "print(MSE(y_train,ytilde))\n", - "ypredict = X_test @ beta\n", - "print(\"Test R2\")\n", - "print(R2(y_test,ypredict))\n", - "print(\"Test MSE\")\n", - "print(MSE(y_test,ypredict))" + "Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits." ] }, { "cell_type": "markdown", - "id": "9e8ca2ba", - "metadata": {}, + "id": "e921f4b5", + "metadata": { + "editable": true + }, "source": [ - "" - ] - }, - { - "cell_type": "markdown", - "id": "843419d0", - "metadata": {}, - "source": [ - "## Exercise 3: Normalizing our data\n", + "## Exercise 3: Split data in test and training data\n", "\n", - "A much used approach before starting to train the data is to preprocess our\n", - "data. Normally the data may need a rescaling and/or may be sensitive\n", - "to extreme values. Scaling the data renders our inputs much more\n", - "suitable for the algorithms we want to employ.\n", - "\n", - "**Scikit-Learn** has several functions which allow us to rescale the\n", - "data, normally resulting in much better results in terms of various\n", - "accuracy scores. The **StandardScaler** function in **Scikit-Learn**\n", - "ensures that for each feature/predictor we study the mean value is\n", - "zero and the variance is one (every column in the design/feature\n", - "matrix). This scaling has the drawback that it does not ensure that\n", - "we have a particular maximum or minimum in our data set. Another\n", - "function included in **Scikit-Learn** is the **MinMaxScaler** which\n", - "ensures that all features are exactly between $0$ and $1$. The\n", - "\n", - "The **Normalizer** scales each data\n", - "point such that the feature vector has a euclidean length of one. In other words, it\n", - "projects a data point on the circle (or sphere in the case of higher dimensions) with a\n", - "radius of 1. This means every data point is scaled by a different number (by the\n", - "inverse of it’s length).\n", - "This normalization is often used when only the direction (or angle) of the data matters,\n", - "not the length of the feature vector.\n", - "\n", - "The **RobustScaler** works similarly to the StandardScaler in that it\n", - "ensures statistical properties for each feature that guarantee that\n", - "they are on the same scale. However, the RobustScaler uses the median\n", - "and quartiles, instead of mean and variance. This makes the\n", - "RobustScaler ignore data points that are very different from the rest\n", - "(like measurement errors). These odd data points are also called\n", - "outliers, and might often lead to trouble for other scaling\n", - "techniques.\n", - "\n", - "It also common to split the data in a **training** set and a **testing** set. A typical split is to use $80\\%$ of the data for training and the rest\n", - "for testing. This can be done as follows with our design matrix $\\boldsymbol{X}$ and data $\\boldsymbol{y}$ (remember to import **scikit-learn**)" - ] - }, - { - "cell_type": "code", - "execution_count": 50, - "id": "0d678956", - "metadata": {}, - "outputs": [], - "source": [ - "# split in training and test data\n", - "X_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.2)" - ] - }, - { - "cell_type": "markdown", - "id": "e735e6b4", - "metadata": {}, - "source": [ - "Then we can use the standard scaler to scale our data as" - ] - }, - { - "cell_type": "code", - "execution_count": 51, - "id": "180127f6", - "metadata": {}, - "outputs": [], - "source": [ - "scaler = StandardScaler()\n", - "scaler.fit(X_train)\n", - "X_train_scaled = scaler.transform(X_train)\n", - "X_test_scaled = scaler.transform(X_test)" - ] - }, - { - "cell_type": "markdown", - "id": "85ca394f", - "metadata": {}, - "source": [ "In this exercise we want you to to compute the MSE for the training\n", "data and the test data as function of the complexity of a polynomial,\n", - "that is the degree of a given polynomial. We want you also to compute the $R2$ score as function of the complexity of the model for both training data and test data. You should also run the calculation with and without scaling. \n", + "that is the degree of a given polynomial.\n", "\n", - "One of \n", - "the aims is to reproduce Figure 2.11 of [Hastie et al](https://github.com/CompPhysics/MLErasmus/blob/master/doc/Textbooks/elementsstat.pdf).\n", + "The aim is to reproduce Figure 2.11 of [Hastie et al](https://github.com/CompPhysics/MLErasmus/blob/master/doc/Textbooks/elementsstat.pdf).\n", "\n", - "Our data is defined by $x\\in [-3,3]$ with a total of for example $100$ data points." + "Our data is defined by $x\\in [-3,3]$ with a total of for example $100$ data points. You should try to vary $n$ in your analysis." ] }, { "cell_type": "code", - "execution_count": 52, - "id": "da7e6454", - "metadata": {}, + "execution_count": 47, + "id": "366035cc", + "metadata": { + "collapsed": false, + "editable": true + }, "outputs": [], "source": [ "np.random.seed()\n", "n = 100\n", - "maxdegree = 14\n", "# Make data set.\n", "x = np.linspace(-3, 3, n).reshape(-1, 1)\n", "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)" @@ -4644,59 +4859,49 @@ }, { "cell_type": "markdown", - "id": "ded6b167", - "metadata": {}, + "id": "520218c7", + "metadata": { + "editable": true + }, "source": [ "where $y$ is the function we want to fit with a given polynomial." ] }, { "cell_type": "markdown", - "id": "980f436a", - "metadata": {}, + "id": "890cb990", + "metadata": { + "editable": true + }, "source": [ "**a)**\n", - "Write a first code which sets up a design matrix $X$ defined by a fifth-order polynomial. Scale your data and split it in training and test data." + "Write a first code which sets up a design matrix $X$ defined by a fifth-order polynomial and split your data set in training and test data." ] }, { "cell_type": "markdown", - "id": "e6fa80d2", - "metadata": {}, + "id": "c9d625cc", + "metadata": { + "editable": true + }, "source": [ "**b)**\n", - "Perform an ordinary least squares and compute the means squared error and the $R2$ factor for the training data and the test data, with and without scaling." + "Perform an ordinary least squares fitting and compute the means squared error for the training data and the test data." ] }, { "cell_type": "markdown", - "id": "d1f9e0b8", - "metadata": {}, + "id": "068e3fc7", + "metadata": { + "editable": true + }, "source": [ "**c)**\n", - "Add now a model which allows you to make polynomials up to degree $15$. Perform a standard OLS fitting of the training data and compute the MSE and $R2$ for the training and test data and plot both test and training data MSE and $R2$ as functions of the polynomial degree. Compare what you see with Figure 2.11 of Hastie et al. Comment your results. For which polynomial degree do you find an optimal MSE (smallest value)?" + "Add now a model which allows you to make polynomials up to degree $15$. Perform a standard OLS fitting of the training data and compute the MSE for the training and test data and plot both test and training data MSE as functions of the polynomial degree. Compare what you see with Figure 2.11 of Hastie et al. Comment your results. For which polynomial degree do you find an optimal MSE (smallest value)?" ] } ], - "metadata": { - "kernelspec": { - "display_name": "Python 3 (ipykernel)", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.9.16" - } - }, + "metadata": {}, "nbformat": 4, "nbformat_minor": 5 } diff --git a/doc/src/week34/week34.do.txt b/doc/src/week34/week34.do.txt index 78471bab3..fa8458260 100644 --- a/doc/src/week34/week34.do.txt +++ b/doc/src/week34/week34.do.txt @@ -2463,6 +2463,7 @@ print(MSE(y_test,ypredict)) !split ===== Exercises ===== + Here are three possible exercises for week 34 @@ -2569,14 +2570,12 @@ In this exercise we want you to to compute the MSE for the training data and the test data as function of the complexity of a polynomial, that is the degree of a given polynomial. -One of -the aims is to reproduce Figure 2.11 of "Hastie et al":"https://github.com/CompPhysics/MLErasmus/blob/master/doc/Textbooks/elementsstat.pdf". +The aim is to reproduce Figure 2.11 of "Hastie et al":"https://github.com/CompPhysics/MLErasmus/blob/master/doc/Textbooks/elementsstat.pdf". -Our data is defined by $x\in [-3,3]$ with a total of for example $100$ data points. +Our data is defined by $x\in [-3,3]$ with a total of for example $100$ data points. You should try to vary $n$ in your analysis. !bc pycod np.random.seed() n = 100 -maxdegree = 14 # Make data set. x = np.linspace(-3, 3, n).reshape(-1, 1) y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape) @@ -2587,7 +2586,7 @@ Write a first code which sets up a design matrix $X$ defined by a fifth-order po !esubex !bsubex -Perform an ordinary least squares and compute the means squared error for the training data and the test data. +Perform an ordinary least squares fitting and compute the means squared error for the training data and the test data. !esubex !bsubex