From c1a07a1b9fdbb5d0a64d3e5fb46aade69e7fddb3 Mon Sep 17 00:00:00 2001 From: Morten Hjorth-Jensen Date: Sun, 14 Sep 2025 20:26:06 +0200 Subject: [PATCH] beta--> theta --- .../_build/.doctrees/environment.pickle | Bin 334017 -> 334020 bytes .../_build/.doctrees/week38.doctree | Bin 198303 -> 198578 bytes .../_build/html/_sources/week38.ipynb | 394 +++++++++--------- doc/LectureNotes/_build/html/searchindex.js | 2 +- doc/LectureNotes/_build/html/week38.html | 154 +++---- .../_build/jupyter_execute/week38.ipynb | 394 +++++++++--------- doc/LectureNotes/week38.ipynb | 394 +++++++++--------- doc/pub/week38/html/._week38-bs000.html | 6 +- doc/pub/week38/html/._week38-bs001.html | 6 +- doc/pub/week38/html/._week38-bs002.html | 6 +- doc/pub/week38/html/._week38-bs003.html | 10 +- doc/pub/week38/html/._week38-bs004.html | 8 +- doc/pub/week38/html/._week38-bs005.html | 26 +- doc/pub/week38/html/._week38-bs006.html | 52 +-- doc/pub/week38/html/._week38-bs007.html | 12 +- doc/pub/week38/html/._week38-bs008.html | 16 +- doc/pub/week38/html/._week38-bs009.html | 8 +- doc/pub/week38/html/._week38-bs010.html | 18 +- doc/pub/week38/html/._week38-bs011.html | 6 +- doc/pub/week38/html/._week38-bs012.html | 6 +- doc/pub/week38/html/._week38-bs013.html | 6 +- doc/pub/week38/html/._week38-bs014.html | 6 +- doc/pub/week38/html/._week38-bs015.html | 6 +- doc/pub/week38/html/._week38-bs016.html | 6 +- doc/pub/week38/html/._week38-bs017.html | 6 +- doc/pub/week38/html/._week38-bs018.html | 6 +- doc/pub/week38/html/._week38-bs019.html | 6 +- doc/pub/week38/html/._week38-bs020.html | 6 +- doc/pub/week38/html/._week38-bs021.html | 6 +- doc/pub/week38/html/._week38-bs022.html | 6 +- doc/pub/week38/html/._week38-bs023.html | 14 +- doc/pub/week38/html/._week38-bs024.html | 12 +- doc/pub/week38/html/._week38-bs025.html | 12 +- doc/pub/week38/html/._week38-bs026.html | 14 +- doc/pub/week38/html/._week38-bs027.html | 6 +- doc/pub/week38/html/._week38-bs028.html | 16 +- doc/pub/week38/html/._week38-bs029.html | 6 +- doc/pub/week38/html/._week38-bs030.html | 6 +- doc/pub/week38/html/._week38-bs031.html | 14 +- doc/pub/week38/html/._week38-bs032.html | 6 +- doc/pub/week38/html/._week38-bs033.html | 6 +- doc/pub/week38/html/._week38-bs034.html | 6 +- doc/pub/week38/html/._week38-bs035.html | 6 +- doc/pub/week38/html/._week38-bs036.html | 6 +- doc/pub/week38/html/._week38-bs037.html | 6 +- doc/pub/week38/html/._week38-bs038.html | 6 +- doc/pub/week38/html/._week38-bs039.html | 6 +- doc/pub/week38/html/._week38-bs040.html | 6 +- doc/pub/week38/html/._week38-bs041.html | 6 +- doc/pub/week38/html/._week38-bs042.html | 6 +- doc/pub/week38/html/week38-bs.html | 6 +- doc/pub/week38/html/week38-reveal.html | 148 +++---- doc/pub/week38/html/week38-solarized.html | 152 +++---- doc/pub/week38/html/week38.html | 152 +++---- doc/pub/week38/ipynb/ipynb-week38-src.tar.gz | Bin 1022756 -> 1022756 bytes doc/pub/week38/ipynb/week38.ipynb | 394 +++++++++--------- doc/src/week38/week38.do.txt | 148 +++---- 57 files changed, 1372 insertions(+), 1372 deletions(-) diff --git a/doc/LectureNotes/_build/.doctrees/environment.pickle b/doc/LectureNotes/_build/.doctrees/environment.pickle index fb3966573864677456d699c7b6b0ba91816f91ab..bf6b9921085fc1b1af1b707d4a656dc1f040a3d9 100644 GIT binary patch delta 11515 zcmZWvd0kCG$1m* z!awzzc={~!Pi|8WHx#&La$-!G2o#6B@}ozgq?tOfJQqIZf#zj>96qQ%yKNaB>0{Pw zQPx~IFw#tG=VulUj4(q6`Q_NKmy%?jI^#zcI4UC@Xa$a^M_=Rxd1BAQHIN(NaThf% zjjWJY5#dn{CRLB;Ni`fSBRD zbHrnTesa{q0)6AC?~?;1ddJZKi;`F*U?H$*hy{AZ(Flu3EE;3sheZ=CqOoX-1$xKv zIu_^`M>8zY8;<5!pa&c+ut38*T4I5wZtG};5KZhz#KIqo)>xFnq74>kWk*{q(6)|t zSfE86?Xie)j~e1{kRe^R{=v1x@s_@1Oqb2^s5Vh(P?ydAq7G4LRF`eomAXWsVO_T7 zkK&0!Iu>$Q{ODGYyV3zyF!P`+V-`UfYMz5K!h8s2Ni$?j zEEDedF&!E6F?Tw$%*SK*vq<+&M+ZoZc}+04qc%Q2CxshK*tlX)P= zF0Ypnfu>_#d6P})WbT`8mqS~@%HIa!kNI}Fe;})5mY;gxJe^@TYiDJd$y4kyeFd9n zHl5bV44G^PMcvj`%E8gme01EH>R&NC+FJD)iY+AbEnzOR@tFu=b2s2;CY1Q z2gi&c^Xkk*v*N4-vwL<|GkRu%OmD-3%p2K>=Bk+q=9{y+npJWv?Jg5QaC=sw`F=Kx zo6Vb;IXQ`@f0kYLk-Vc>Gq;mjm}NItcJg zPcsM4vCHnCi2ml}1wE`(JJlCb2Bq43M3idRBBE5g9TBD4y@)8)o&T>3+W`@!+CGRV)s8?!sWu%ErP^#nlxkgw zDAh_tlxjajM5%TsBHYA|1BfWqeuW4(Z(qmv2r1RxK}4zc2_j0htRx^xwLyp|)s{m< zskS;I+@OwzfF>^ynQ3s0$}XtxDDN)stit5OG#)AwN3oLbev4`n873>VhJCZ$HQXZI zm6td)dVguYR*5fXM7jGd3pcQXF!!wd30mca8OkkQrFX+MdSk^JE&Y9E3yUZBlQS^m0ZTJTw8Udo-g!L7+5Jy?J{q;P{KW9sq% z_qmOWb+hrC-ue(v{AXK88mGn+NKeHIKvlU#YO zP^0eu0#_nWz8u4R-G6=Snf?3=;7U^VxtOZa+)JJb*)M;hd3Ua?XYz+byixvxrjY9r z%_cu|kVS`i_`AEIzm-2UWk`j(Z(U!g(Ttn3EF|NnL)s3ym8}J*v)PRsE$b zw>s#4ccbRneuK49r+N2cv~>T2ewtVP(R7X6k0Az8Pd@C+eB76x%+P{SPh0C|pZ)Qf z=FR>KjPvDLh>G_zYnuC?caZTvdLkqC??w#CF!$&e^ECSVC3qZpY9(XfXoArOyQZPd zhjru@cestswYW0%Ps3j^KL);s1`%qxKRc=$j0U016v^1 z4BqLhvcuU7W>$F~BM*l0UiesN2FmRNE*9@j>vLX!fy&OUkl68opJY zMQ}?Yy8`R0xp!jOhZ?S_#KveCWoHJLci!b`-e;HGS(SB=u^V&(QDN2Cn+)kNl@*8M zA(Ut5a$h<0TO&rEJHq_bkQ!`|mejpsM9WhNEI>VajlHFrUtBjL)P!0LqMdqrR-3KU zqANcduc`BO*#a#}t>+P`*amD4aU*2V=SHN0USTKG_};Cd`ZQv&9f`S7iFT@C69y4T zbom3KuKG_?bPtU>zZrx0AlR)1gFs*KZg_EN*QlNjs>tAmaVqL<+2PR!J>TW1fHi)V0u^`Z;Q(4x8B*eMMm z30~E(OD}d)!^Ga~k_9Ee0M+m<2Kj(Sc+{6cYQDKsq|r%377PGCK0G3p6$5CbWAVt8NKjU%;{~ z;n9U!s3t99%Pd~_Vz$LX=a#U>R*N#rU{g}Z+w<5?3r);t(U$bR0?$-lE@zy}M=inc zaPrzCulM!%z0FG2k|7ikT2v%6lAYBq%y#_ZP!Ph=uwKzdJ^Be9beK)}7pTCZE z(DLi+Sv51abLbWeqkf5lW zn>!g~ErL_sIHf-#Sj@ZkSs6^(8>h9Eg}cD<2ckU-^WDvQF{H!ff$c_PwO+9inr^lW zc2%uCY^+7Ak3MIEHH_Me1463^wQCL09PqU{Qc07Z7Y)vVi1_$66+4E~y#)2STM!COc&0wky;uaRkQeNZMehpW5TZ272aLm0 z6_r#SgTs-5V3FWpJyei8-V_bY3b*{!+rAt^j~bOP$=7J{gAYY(73k05p+cfcfv9zS zBRpRMgZOk=wAEvq4}v(PBcQ|7tzZt%$Nzi^4{PfMCrHAhzJuIx5<)n)5)W5{!+33m zbeLKb&cSd1BV_j~yrPPY1e?t%A#Pc={1MR*pmFCpayJ zFY`30%S)-aSU!o#2}2E<2)$j&xJodSyyZryf|ilp#U(96Xemoam>4M~y`Y?g`!C*Hs zMN$Kv04IpSV7H|SJcnR{N^HboLlGCw(w}HJvnhw?7jZ>1ZiCYXHUrzhiI<>i$yaNb zl8B>@4CT4_0@+6nF6|ldZW|uWkPefu*vhI=J6ti+>f`ntPO=0$CviA$6I{`OKeC`X zep7(j*ohzEmSjd34)IJ~!2>)>HSNaXC`OHDcjrNtmIPx3sI5IY96Lxgq?;jR;u#UH zYP{*um%xlt&fcDpB&;<+z55nl!sN<|u!!~jU=c;}o<)@J&&xo##lxhN24J`mygZQi zv7oiIk%KXqP-Hpt!VrE!tL_-a8*2D;ILC7;_Ej~7@7LVlNAbNH9vQ!6}J-0hJ*pJF5#8Dk9?TTtEl4>cnv1k=0c}Dop;i3@I#5wxGw|fPz$&j>OeO2$3e6mHWtEkJfPWhZquU1*EVQx!vW0k6_Q{uJWX> z#~<;2a7bGP%?Evq0Yk9%77kA;f`_;A-5R>Kb9m$s_l+WcOv9K@Fc69R%*|J6xM3Hj zGUC3aFsl%3u!q;+7M$<}W$8bv#eCIf3T7pR@ zIUJ)1etU|$G)()7!xNmiL1+0U4NrWHOCc_t5a5|X(EnQwkF!-rAY|dQlSAel_lE3u zydpz7Oucat?MU#-C477#TeyG3L2NJ@LEO5P7|^$HBvHWBLUePCY;I<U-{?}ow zgSWR*qJ}z_Y77KNHWZPn(F9`}Y^gdTS{+X_MrhsU6O9xL$_knAs=9BI0ehIb0UBZw zp(QdW8)?*6gepDNJMOkD1EL)UD{UpjPS2VOpG*qpZ z7~olIRk+lEZB4M&GGm#B@TqB+1?7peMn`om-+(Ofuz`q>qs|!>RnG#WrIvs9jS;I3 zEjM~;+Vg5#NA;IR7tOqP-l(JQtS~xiddhe3)|$G?D}Q+rPF@|7-+;`outF0RhIB}3_ z;!UHjYF23U*Zp7n(P*lE+~{4@xZAL(w>BAX>8=ibY`__jn!*cpBYE`>oM85B@y-`s ztz+c$+rWHdn*mvj#)sE~)f$d3GTzlN@)NIBce@S9gQN?u4pTKey4&l*@t+y6xk(}L zb7MTg2zfnP*j3RNMrTgOQ6c+{x3yN$0b`*B<-NN`eKq5dk)-LY-;C<&`Qd+>w5j^= zsK_24|j8mWN)rxH(|c@Lai`w~nO1+vtA zBUpjs;dK}X0+&y>1n6uD&aSC`yK1!6%IhH+r2JsCvy|a0Ne0a2Bz*2DG-|7~o8WHh z1-_BY#7Rad@H34YCc(o^RIQ&OVrX3SvH~xUvPiINBjx%x7>hap+fK6(SQTEBacVHr zqZYdLz&K=~-H&wls?~pRE~t0(iLuB+L!RozD0rPovv}|c%$kAv`7h5H6Q3LK4FIhH zuQ}neCKI75?uF6D;>~zzd{L5@U?uP;E8R7cmd_U%G?YIR*_IN0S=+$HSZfq`M*;)j zY_|Fct8dlPCKg(}izNkoUqGqDUyQa4^;3WtM?Cyq6Q!n=5?zptP?v+m0!uKovuLKa zg$NxJwZguQ_x*wA zs3zuE{k@12g%;XX18soTs8droEVQ|nXSB+7#4yW3vaU$Aq>=SJdhgU1LoBIC@K`** zp_p&!)o83urw%m{@X;;p7N-vnGWT^+BwO5uW9*yF1RS~mhN`^g(1QhITEa9fSky|u z*_ae2w}vSblr3zcw(8p!7HHAml@u)$YcC-GqpBS1FXGg!Hw5HxYI-gS>}ZM7dvhOI zts4(lqdS6+E#}1_`11FplYn$d`a?R4kF}^~nCPa)bQOa&ef@VxwT;-obv#jEvV6a*k~v8VZCJoZq+0q(NY zSB(>JPREw{2YqC`0k62N$9pCq^Nw?&E=>^d6sOK7r+b$aHA%o>fVeV4{HgS1p*EO z#NEG86ls{VNWfVSx$?!|aN4T6Sim`e==2*rTCH6o;KW9)eBKst+#vYzGOtNmz2hC> zM81I6XHw`};Dx^|7jXV2?glB~C~GOKgj{36hpPk}#fZCjHDnBe5~46#4SnC^ep#+0 zZ>)mWip*DX9|c?6vnKln@wx8h{)ghEh9@=(cn_eymu(VotU)Mi{UPeBq|KtQ78Pyr z&LDZ4CxT>-%ZO8dZTGH!ND+EE>H?gd;nH=ze0Tx2XKA;y!yMCOi_W}^oY$E* zm+rk{Eot*53G&kCB391dCqm?JU3iq*vJVoB)oZx}Vx5L(4hlGJ&{&y=F>4THM{(Z} zoO?{9YbcHjc!MTx;R#POw7<>*BTqr|!6#+*m;-+5{t1KwNV?L_l zbpdB-679Vq;Gjxy;Ex#51RvhQ*hDCoCb0@K@eVvUH~oaRCV9ji0Y_zmdw;<_L$J?X zNaGg#_E&T{aYxIOaVa@Rc+?e+SeJr)PF?(rwMzo?n>DQ+5q5OoENuIZD*>5e|R z0$tE2htlnQau8k3C;QR8d~z^d$|r}=jeN2%UB@T;(_K6?$BTH7;VnF9;uSnp_-C8s zNV;DqrH8_QGUYi~!+#DRRE!EK zM#U7PVvA9=icxinQSrs7`d$kDCz0n!@RCTiE=IL2Mzt?SB^9GO7Ne4jQC*5r-Cj|R zV89-)NT7NZqk0#khI=U4z|W^Hd>k6#=TnbU5D&tCXqG4e75>WvD*THHRQUfAsPNAv zP~ksIpu)eEK!yJ)feQan0u=@^R2aliVGu)wK@1iCfdnf2_Xt!N#86=nLq$OhHU4D; zD*V3)R2WH6VGu)wK@1fJF;w^m5vVYTp~4`B3WFFb{7(o}_=gavFo>awk&^>`I?+0v zv`iiD|$}pO|J(9+_FcsK7rtJ9*;BF&W7j zq;K9H7!r)FQ^$-So;upqO(S#JpvY#8iV8}POwG!GXN?&TGH?B~`VBykkvTkbOh)FI zks1H#G}fFpC@dEq<{DnJ|ERz?-Muvg#f*_D$w^tM$-vG`&dwZ>mNxdk{X#chnaRUJ zbs`8|okMeL?l9qCJ;Jo~OgZ)+QM#sE3m*9YrVi$F4grSco7>FAEsxv^a`-?jpL&CVXhk3X~MGt!@ zXqhwHRp22WX8o2Gxjo9fH6+YDHYCIhc*8q43_J0n_^CTCsmN|a+SiH_SBJjH^K-?P zi|d74AD2U^aYbYWyo?x^YA~s~TuQ3p$nwqfa|Ho0E18?#sL|4cL=?8Mt{x&^@|w7K z{x5lM%lVk4hgQl(2ivP-5sgI+EYPd=npmJC?X|E#-`QWm0{vyLjYR|&b+8D);#DlF zU{MzfPb}(Tfu6I+V1eGU*T(`qWN+Y+3ne98s6Rt3p8~*duxPfVtX7Gm9S`oMHwvG zVu4n+x5EN$Yj2MQTGZYF3!maq!%7=uNT+SgXKj~%)XI~L>9oafZch{%)M=}`rvp)F zRHv=lf{sL?VV$pBsI#s$qzU5P>iJ8kD8yAg#(cG^xCTNE1FY5OU!2gy8hiTto( zPa@IcPFuiJi$beAZJkaekPI#Fv^`nbizqa{)3)$>AEM9!PTLm^`x1ptaN2h7vM6+f z)7Ia=AIZ=eFtjcf6@`v;+V(H9Nc4)+HuQ!?p=X@71LgWtL-dZ*mXu;q=pm;q>-GSW zp_iPtbyEfsg`RTS);lcCA$gT#Ybml*dxq<(UFkLVn(-y zT$T>lFbkm+<}N79n-`$0Y(9cA&<4tk!DrU@t^3+e9W+kbzo{;Cf+tr&51Sxwly+mW^^@6XGfbqXE@D( ztY~v(W|ld7O0-NV%bJ;~SzXOLv!dnA3c}a4&j~Zxq^{=1>}Z+T8s_unq^>aMz^o{F zXb7ufhE4w2Je(VCzM7qFX5~f8^kr;<*=$NzGjw{ixoOHmGdnZd>^F6^89yo7yft;X z*>GyKxoX;U^X;5QX1D1*%*rzw$#+f~e&)65app%LY@IXAteP7whqdLA=B=DKGYmBQ z%;*88sWPLX>^`3RnRjQznd@f2({gxIGdDNR+?N$C`$^u#tew}@oH8xiT$Wd8p3jJu z-H!1vbM(wObLgyS^W4lL^Y^*Ya^58#Wv-tUXO05m#vNYX44?hPTroXbUQaaonDgeu znf`O4<+ig%OSAgiIP=|^(PsYKH1o=&XqoW77-VM6>t&_b$)1oZD8(*8L@9O+B1*A4 z5K)Rfh=@|`IYg9VuOOln`#mB`u}=|Eisd#ylwt!AQHqT~L@Bn8&7ERjLnfuz&WI?* z_CrJ|b|fN7vFV5?#pWQQ6zfDpDOMt)6uSWtrP$qwD8(K|L@D-TM7U@N*uO?dDfS*B zO0iE7QHo_=fGEZKA)*vp2@$2(nuu^=+8YDPDiE1zaJBZK48YcpFxsc!zU@MHW{n7^~6hg{D@iSIG!19=Isj!1jT~IZM*C%5yV_TfE9| zg=o}a+1pzBba^X_C->(WF~u3H3bY{L?K8UPj|+{K#nsmg(t^cn3$;Ks;eN$&>x(rx zv={R&4%oO}lVSC^Z}Ejq3v{#8_vF6?mAA}fvgKJL4Ho5enSR~~lyRe&SMkPeuQ4Qp ziXU%>NfH{rbCE{=yBlze_j<8I^B(NU*T}KYsZrJaQ#9InU>u>4;+PNTYu?#|MH;0Y z%470TLlIJJ9Ld(ag-1R0Gd3QFnI;B+w984?2=1Snto1U^OxK8=1C&rkgv!=$K&pHD zysN94pKQ@8cRvNQ&}5ch*r-v$=SwUkU!;Jy|NO!==DC-~YZZ^nsf0p`=UxHF*(XEDe%boZ0k7XBjt(X*({z(TpJ(RDAd5JdLK`&e6!{4$PaRtM8uE zXv+61HLCtY1#U@u-G{-Fp4U$>@5Fpa+~$LdT6*~504=Tc7{*R|d!7Vn=~qv4HE;B@ zIL$loyVAT_f5O;*4`JalA&OO4@~{~s;@ak+zdOsrp{_YZK96Nc1{II~XO2eCUx1I1 zrsv_afN_4OdrTDFj1Qg=ogOOuNG^gUN(yY_n{6*UA(s zs52o+Yvu^6uGYd+-ul+4roODlAe=}(w!TZQA{wHpN!8~ABUC}xuy~11 z@5?=8%QxUr{bCucRBHOD2|K3c5r;*L`lT7VhD39kqkRbWXvtvJuw^U__umwmx58!R z+pSq`SPm<|$`NhYBn>yUWf0)Vl?h#7z_=P!`}VAcmgjU}FEo6!6Pux7L>D$y!!2DI zq*Lm)XEzs=3#ReX>Ywf`LyKnjWUyvPYgjLKO~dZ5v)dZR^<`HqC;|GY#{JPH)Zp;| z7R=4HThr(WA&UmH&gRYS)|yqn4rUD)kk;@O48K2->b;>X%|f*j!8)j-o(^}R!y_3q zx%U=NgQG}Y={=csmXYhh|CmsfeKH0l&_QZzHp|qo{!}zLagR@9y)~?ngA0JTAIxB| zkO^MPV@aCZdlp+|!F_XBO)YracUT?R>_AsezsS9%=Xw^R=B{O(wfyEfR^5`T@*CMK&E2|*9nmo7J+@QB zR$Evvt-E#`JFU4>cQ8np^aOw*s>3b@>53kHdpCm|MsRX54k;-R9JXmnJGpT$IBaM+ z*Ni;(vDX>WLGthpqlsFl*hoz`-wP|W&H*;YqSd<}u%Q~3KZsAlR;u_TTtNh99A+)J z1wTH@wrQ^OIQoMWI-SHgAo%qu)=R?{XE4~2i-z)2XV0=xTGZw|d!}KRPhh#3RsT+d zLvFb2{Ta+6!RA__#hkXvxgH4GZ!Hon1zXzu!3Y@z11{T4GVDLlX9hWo!`aNr?s$~_jr<>OYc zYu=jLQ?|bk!@fDdHSCS|akT&)B!B$P2v&dm$VOPS8ux(p(4q(-%B#CSvnN{g!y`Bg4R8I*)>u$JdqYI2J-@MzTIBl&E5ofPruGscs`XziMvG?tjZsW} zy?D-kx8PoH4u&JM49P*r?U>(r-^bbbB!R2lQF%Qo-$@dga(Aa%!|L!x-phUdYHjGufG z$UDn`o)E!#Rd}!(8pK~^NC&Ca!5oZ+Q1+ zVsJ$B{s#foIww+IZ2$wv`wfgnw&>K5H-g;%8yIb2Bc7{aL@bBnD{=QX;qPlWy&1=? z40T0wZsQgN>V8`u%H*0nm$hTt^Whe)Hh18Q zHSE@j!x@A;@rUHNftNv3t@5XaON?4{*Zvz{CdW7KG70{64DB-qJE zZR^e9@Ik5pJq;myOofG0tB*?`oM)6*j=ruZNtmXOdaFN&tkAsy_-)-l-cS0*x}qm+ z5U;?H4pN;4W0(>AW(e;`u#rj{iY`R1<-$wD_!+IWb2x9T;j+Wa9S2JOL_JrFf28DC7W-x(7h?JT8Rf|4G4ix#^FYn z1GagT<}ypj3^-aM9VFq%<1dF~!)S)&^7fj!+zU=;m-Bc}i&jl%VV*#J_33OrR>SzY zF7DD!Mpfl8pMRz$-#gr0{TvBh4R^xytmmawF>0$H7QjMdHZ^!59|5&I^d|oi$SsR_ zTO=P8A}POK49m%;PA}m#QTdU{_hP+iDR0AM#cGgL;IE3s@@5mARQ4_7wILBTffNR_ z3RQiV^DK*2*H)lENL25t;iC#yaoDIWRUvHZ@oN5urk~evO6{*H#*4{_MB ztl=EN*vY>hE7wHxX0rP+KGMv)VGUb;d5oWu-;+{qEuAvCgv|}iK#{B8^EJHB81d7gg2pyBtn2j|RawzW$Ccohp}ChGhnSd##eYVqD8?^>8B$j%hKlY|f7` zV3S5DV@4T5)krda(;}efX?SL|F;~N3V+=g)VcWm####&Ft4MiuGSz?`h(ry?8`CvB znr4jAu*C!;SvQEtG)k+FCK|AksR7UsnKTX(nOR1f7KKbU=4-e;+W-fW7FtC?7EaNk zGt&&nwIo`U04K`TIqt#u8wc5DQfW=L)&LU%xhMyD~do3u>d}?%2*OwTO zD@ZG*$Vk+puPz#q>gZDAbxlJ~^Tc=Cx@wR#x@+cxFM+wSj)@&Z1~H~UmJDQ+SLXm7n%?xn_8;sI|dwviH0Mmr&_qi7;n+4%sQ80<%c&R zQEpprz&V3d<8B-ERP&9-AWdKW)@ZK2-Q=2&9CO!*QvKgE`s=QaY&PInNKI4gvsiiU zo)M@HY;}(}vLXD-F!e4lJ8m~1jiIK3mjai|#_clR(%iD|yA8Up*nmVxy70mP0Na}mesKk&ClCbC? zf1k%uA+3TxF(B;_gy?|fMOqLTQEJ`=qZQHh)vp&ZFi3=fp&&3Ixlk*J2{1v-3ZsQu z^A%c_xDW~Xx&a1(f*^owMk4S(#5VPXu9x8Y+UlokMmv}OI@maqZy2!gl0F0hn42K@ zU-zoO^{~o`3tpe1+k(TJsyg3e03~iU2C4`5jJ{d~HjY)!|DoZi6&PB#Qeb1qn#6^{ zp3v~lL*uB1`yS&QNul*G?$Mlh>K@IoXRbLZc&&l#LAt4b;6Mm||EK$D6aF^xEUtuC zo?uyn@I#*;^-IQhm{(-?BC3Id|t>*#~96bTv znOp|h<{|VGR4bc+1V##%y~Ion?WF}oJ8|#$2ptMmBArr3xO{lUPdF?snb=Jh-EBAL1I9+EiA+s}J?vx4eME7C{Jpc#S~#uRSHZqJWi6@+*~uEB5F( zyC7V^3Lw$l2mz^qU{YlPI}E{VRm21fS{*K1D}vReXjdzXd1wHfTfVObRd8RaXqx+@{u3jI}(W|wGkV%2s~a#4QPjRCRK2I3&lDJ*guE{-&a>NI*K-$h5(3CzjPAu z7Oh}AP0}I=fV%2wR{`mko&r(uo)$p>^iU~1#86FR6v+7zkV(rW2w1tK5AWL$hXnuW zE#Txu5Z;%!XgIc?fD}mFG6Mv>ixXThP$X-}2a7}vGlz(D4dE4E8$#s|6Oc@(S=r&} z1%iu5h>;eQbu(ah@lA5~I6K*`^NCDpeO@d|BmGJ_0 zd*V(?ca2{jIS((-VH4e*FUt^rYpqRL?z#7$?4ISNY+Oy$O`oZ5-4WBpLd|6}#9|FM z6j|^?Xw-(QwNmw?X0-yB~9Q2}U>R4lHuR|1A}pHFtv)>s<=VG3HShj-r7w zI2sPj!7Fj8QTu{bVygvlftFW^Z{vDHk%a$IVwK+-0ec70bn=3w?(%un_2Pi;1^#C_ zqv6?20!}>Ccj0^PwlTlML8{XhF+huUZ^h+DT5!gKL`6@4<5hL_=MJ|a;EV;a2CFFz zuV>fE?IT_PJd`UIaQX*2NG0yUEQ?UWBCVk+@585(Xp<7lG)y=kvNXK>0i+54pdlmD zQj;>KjDu?0kP))uK@ssEW^Om$M(#Z*)=+<}p-lNm)Z-(P<(hCFqIP_Q`G%epd04E` z@Zu2xhXUeeAH)1X(BlNIAcFHx3fQCwdYu+1bX zsbS`Iv0TH*o9^Krxh3Az+{AA&ys3w$cQ8f~O1OjwS1Z25B}byrdjgJg1mT|pSUUs< z-4|muy!<0No4ED|Zrz}VZtliMqP?EgUr)p#t$Y3{t}|*@@C?@tL5Me#(N%tYFx}(F zN6;mHd?4N6$NSOseY`i_-N*aW#eIAL-P*@{(v^LDX}Yh6=6G2TGQ6n=O}wUu3O~e& zFH86H@tiK_p)|# z-a&>J@6b2iy2p2>EBE*|bl(mXUbe?~wfm?Ie~MKeKJng5r-ag(rPSVdk91G?v*Mtq z$LU;rSfTyPMx{zEf-887`lh%&--xAaa7bRo8J?g<9OQXC!>T?R>H}D(IQl+56FHxYv&rqPkZ&0AZ z4^W`OuTP-DPfwu2?@pk?k4~V%AchKq7%B{6s4$43!jDX#!Y@pq!XSnUgBU6bVyN-6 z5~%Q-5~whepu!-A3WFFb3}UG8BNC`Eh@rwDh6;liD*SE)D*R{!Dhy(%d}Ow-M^~Ds zgQn@ASvqKv4w|Ebrs$v}&MxC&aON%N`+M|b wA@T-R9SbRzEH!0Jd!>C4|BxH;{X5v diff --git a/doc/LectureNotes/_build/.doctrees/week38.doctree b/doc/LectureNotes/_build/.doctrees/week38.doctree index 878f112613577eab13ef959d200749d21578cc58..19b54d214291f2fb72acf3a48c2535be0d47a286 100644 GIT binary patch delta 4437 zcmds5dr(y873W+Qlogi8;sX)5D3r(e0F6ks5-4Iy0EHll6PL%zyY51Pg%og89~gox z7&yCr@vRLe1yR8}Q&LM9GBw{D>&;-##ruCiRKE?6033tik6u-vAt0|4RZpB`+NOOZzr z8=j9~n8Y+xm=>c!rJx3msJ1Otsb)_I5^e79-?W>)!8W`5Nr zN+w>IE0XfJB&)#+>1uannPe}!W`|@oQ9chRYs5=u)s$H~X3Yd^?@nS}h(uIV_Y=wL zqdJxC%Kjh6_*)sv2-WGGh9Df!rCGzbp3SLhB^H%?U{rk)-H2{{0uR@}M1O`i+y)a4 z>NY$|AjHe|`W_F%6+dr-O?awwfyKgVYsw8a-fPi~3{>OE<_L@3bTZrlvuAKP#m*#v zdzhuC6@hzSp9U_ySZ>uXLKt`(b9Re&vu5`m)^dRhnpIl0uA)}kyg4&hub(caQY>|A zXwM$FjEQ@npo^N_!FZ@8kp4WjcLYkXu!WXfDgRtcup>Sfv@ef=QLJtavIxw@D1K@5^B0vF)fK_{#)^=9qagu(OmQ^M zBvYcN>=Lz`a*V~pN5mICdE`5~ckgH$KonLTXZ0>&oci3Vm>tnV8*RI zV*fhOQ%aL^r8gL}dgPAyoZsoSGU&mK-e`&p>)U1#h;+aU#Y24>IEm-ati;4Kw0dzT z&P=*^aD!)exx>FO;lfI4$orzkVlNoK989z}{Fv!W`UoI39m{#nf&q+wdx@PQ8mS|` z-)V0=d9fHyV86Klx0`=RwH!4+?_vP#@CP1{hktr+FO}c%{$+4D1tooO-%I(LJT=kZ#v}ws@g+7<(+L6I#1RpYarL zS_JxBi{n~HgB(VJgIDiEE*`smD8=MIFqdp?HXIfl0DuA7qmAEua^C zh-I4^-@&!zXt_oI5Na zHC;d-GdbJ^J8gxHj2j`8tnCK*9BZkQmsQu2iSJ-OxnqJzvcDVh933D3bvNiB)LW&k zFSN~kwzfFMB)g2z6X9!_TE7RA^7)M>2!YrXxk^)7ri+$pRsBrMs>*8B#m!9;MWp>8 zurA_*_%9B^q;G77FcsuhRuz}B5vms|OG+hUE4!YRvdBO$1QNeDfR<^J@&^2d9*ftn zU^B3*q(I?gNAEy(Zj|*JI|IH!GH-)dh~&$tC!!6V~|tcfk{!H#ar{ra5gNfMC&)ecXE1mUnV9}LvM`}^6O9kEk+gd=^b?)j^k#MmlvnvMXI!h-e7rC7 zIzXcMZvgV753|UwvPyQSr3`=OBD_dC1DGY`PAe#;s-f`l_!QX}z~s49*?F0gaR)d= rdIK5y*2sr}OrvespW^ZY_xJaGO%FZ9p5^vaQ;eHi{m|>~4?O-2lTMdj delta 4072 zcmdT{dr*|u73cTy3ag^LjLL3Yg`_MZLE|fJ5Qe z^@flt`p;hboJN5Xc0S<;!Zb~p>jw=}&xj6o;4Chn#3K2E8D~RUMHy3eVoMWA-E0Mz{FI^VkK*y(7JeoDVY$o3WUh~8YD zOmuu&64AA3Gl=e$=+(3(v{t335;dmR{wyskQ?D{&B-_1e0a^KK)h42;8P!S8q)+?kMo6V+=>7#1UnS1oz1_lDKAG5#3SO(n$ zvhrlj-6$!m#6G6_ZeFGg9X=4+J}xIrKoX13olZ9Ox!FH6vVZ5cV*zW?zd%+V>NP~$ zOyi%HJZiGkuk{VlAtS_HQ?Z`|!9oh>vwa1lh}BfkM)akk2%?dObME!`vUOX2o|`}S z^*7w>$s5+W*K^&gViKa1aVJ#NAj`nF~FJlOo#(w0tDghs|9Tkq#a2{?yPvm$x%tCFuPQ1(wPk#Or^_GR#@EEOopQXZQEB0xXvuQ z$C7HmTkW#-dDdVxqJ}AUl@f60uK7dC@C$aLT1kdGswWeDQhoO+XUop)Bd|ZI zPJv&EqtRd8lgw&ro5)|%o)f}QDmSiv4ltEv*y5#bUavDmX4M;1S@l)Z_3VmvB0Far zL1r)8W(A;?VPBEIR3x@MV;tMF?>32w8mz!`EN!29q$F}zOyUtB<-q&G;DAG2zeu|N04 zmO|`jjV+uM=e4dLQjC``eBzH^v6_#!l5+T;E@6P7|J^4KfRFzL?R(>;2+4D{s5MVFAdHPw~Z2T_BR^~%jkp5WvV&$2CVahh#?-VSP;{M1{}D5dl%agTB3c&ldZ@e zLI>a-R-C{KtlBgtdJyBTXOIN!8Q1-y5dX0Ye!w06uo87tbQ&Y#l`5oq(f?W57KIpG zg9Kkh@T5A#P{R3j*g`+|>OgIfk7jtM{}zfLF_@torZsI`@ zth+W^&KK*BA>7wH=;uxN!Ji7#_5o-Zi+4Z3FX5f$g0pB0fS&7GG3(iAkFcWE>vTq= zK36Rl@MWzi_R2WF-->FU(v2Wh&nd}{b!8eNb0zM-faSn#{?9hh*8!h-k^ar$){A)P zl^^k7I|AU1qbry2ZWye*s1q~ELt`iCtuB7vNhjY6b*}6Jj|z50PThdcfh!?7TkPt> zPG3yoYi}VV(1FU1jpo<75zD{3g)`*&;B6eFpLuscpB%jM4%)f;EP|9yo4sWkC37Yg zccW48B2IDsU-342162$hD6z%=euENt;rhVe@KYopHb6K&3@W~Ogh0cbe>*}brvl~$ z3&)1iY{kS;!XJQ{(nl?y9V#gMe%8qZ;oTv$TQMSB@CEkr r<_LlA2fq~|l=Qsrs$5Kwm{%T*eBGw;@!4p5>@Q6hAMqS~AN&6g0Y>xp diff --git a/doc/LectureNotes/_build/html/_sources/week38.ipynb b/doc/LectureNotes/_build/html/_sources/week38.ipynb index c474d1c83..544d286d0 100644 --- a/doc/LectureNotes/_build/html/_sources/week38.ipynb +++ b/doc/LectureNotes/_build/html/_sources/week38.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "3896fc4b", + "id": "169923b3", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "ec008b77", + "id": "47013ee7", "metadata": { "editable": true }, @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "997c17fc", + "id": "d1fb5464", "metadata": { "editable": true }, @@ -47,7 +47,7 @@ }, { "cell_type": "markdown", - "id": "893cd04d", + "id": "1a34a7ce", "metadata": { "editable": true }, @@ -68,7 +68,7 @@ }, { "cell_type": "markdown", - "id": "115e99b5", + "id": "c6f56f83", "metadata": { "editable": true }, @@ -81,7 +81,7 @@ "particular, we will focus on what the regularization terms can result\n", "in. We will amongst other things show that the regularization\n", "parameter can reduce considerably the variance of the parameters\n", - "$\\beta$.\n", + "$\\theta$.\n", "\n", "On of the advantages of doing linear regression is that we actually end up with\n", "analytical expressions for several statistical quantities. \n", @@ -96,7 +96,7 @@ }, { "cell_type": "markdown", - "id": "857956f2", + "id": "3ece0c04", "metadata": { "editable": true }, @@ -112,7 +112,7 @@ }, { "cell_type": "markdown", - "id": "074e7f9c", + "id": "d36cf6db", "metadata": { "editable": true }, @@ -120,7 +120,7 @@ "The randomness of $\\varepsilon_i$ implies that\n", "$\\mathbf{y}_i$ is also a random variable. In particular,\n", "$\\mathbf{y}_i$ is normally distributed, because $\\varepsilon_i \\sim\n", - "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\beta}$ is a\n", + "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\theta}$ is a\n", "non-random scalar. To specify the parameters of the distribution of\n", "$\\mathbf{y}_i$ we need to calculate its first two moments. \n", "\n", @@ -131,7 +131,7 @@ }, { "cell_type": "markdown", - "id": "0220f12c", + "id": "5903d2be", "metadata": { "editable": true }, @@ -145,7 +145,7 @@ }, { "cell_type": "markdown", - "id": "fe96cf24", + "id": "37f4e199", "metadata": { "editable": true }, @@ -157,7 +157,7 @@ }, { "cell_type": "markdown", - "id": "b308e3f9", + "id": "7a7b084f", "metadata": { "editable": true }, @@ -168,19 +168,19 @@ }, { "cell_type": "markdown", - "id": "a2d28d54", + "id": "1ec511a1", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta}.\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "501243ea", + "id": "ffe76935", "metadata": { "editable": true }, @@ -192,7 +192,7 @@ }, { "cell_type": "markdown", - "id": "a0735fd3", + "id": "e569274a", "metadata": { "editable": true }, @@ -200,15 +200,15 @@ "$$\n", "\\begin{align*} \n", "\\mathbb{E}(y_i) & =\n", - "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}) + \\mathbb{E}(\\varepsilon_i)\n", - "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\beta, \n", + "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}) + \\mathbb{E}(\\varepsilon_i)\n", + "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\theta, \n", "\\end{align*}\n", "$$" ] }, { "cell_type": "markdown", - "id": "105564c8", + "id": "c4dd2623", "metadata": { "editable": true }, @@ -219,7 +219,7 @@ }, { "cell_type": "markdown", - "id": "e484179a", + "id": "df2f8936", "metadata": { "editable": true }, @@ -228,12 +228,12 @@ "\\begin{align*} \\mbox{Var}(y_i) & = \\mathbb{E} \\{ [y_i\n", "- \\mathbb{E}(y_i)]^2 \\} \\, \\, \\, = \\, \\, \\, \\mathbb{E} ( y_i^2 ) -\n", "[\\mathbb{E}(y_i)]^2 \\\\ & = \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\,\n", - "\\beta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \\\\ &\n", - "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2 \\varepsilon_i\n", - "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", - "\\ast} \\, \\beta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2\n", - "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} +\n", - "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \n", + "\\theta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \\\\ &\n", + "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2 \\varepsilon_i\n", + "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", + "\\ast} \\, \\theta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2\n", + "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} +\n", + "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \n", "\\\\ & = \\mathbb{E}(\\varepsilon_i^2 ) \\, \\, \\, = \\, \\, \\,\n", "\\mbox{Var}(\\varepsilon_i) \\, \\, \\, = \\, \\, \\, \\sigma^2. \n", "\\end{align*}\n", @@ -242,42 +242,42 @@ }, { "cell_type": "markdown", - "id": "028be304", + "id": "0a3e5956", "metadata": { "editable": true }, "source": [ - "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", - "mean value $\\boldsymbol{X}\\boldsymbol{\\beta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." + "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", + "mean value $\\boldsymbol{X}\\boldsymbol{\\theta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." ] }, { "cell_type": "markdown", - "id": "71256ba5", + "id": "973e45a3", "metadata": { "editable": true }, "source": [ - "## Expectation value and variance for $\\boldsymbol{\\beta}$\n", + "## Expectation value and variance for $\\boldsymbol{\\theta}$\n", "\n", - "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\beta}}$ we can evaluate the expectation value" + "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\theta}}$ we can evaluate the expectation value" ] }, { "cell_type": "markdown", - "id": "236fd8a6", + "id": "ff486d1e", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E}(\\boldsymbol{\\hat{\\beta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\beta}=\\boldsymbol{\\beta}.\n", + "\\mathbb{E}(\\boldsymbol{\\hat{\\theta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\theta}=\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "8d2e261d", + "id": "e4307815", "metadata": { "editable": true }, @@ -286,35 +286,35 @@ "\n", "We can also calculate the variance\n", "\n", - "The variance of the optimal value $\\boldsymbol{\\hat{\\beta}}$ is" + "The variance of the optimal value $\\boldsymbol{\\hat{\\theta}}$ is" ] }, { "cell_type": "markdown", - "id": "2d8ebdef", + "id": "490b2cbf", "metadata": { "editable": true }, "source": [ "$$\n", "\\begin{eqnarray*}\n", - "\\mbox{Var}(\\boldsymbol{\\hat{\\beta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})] [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})]^{T} \\}\n", + "\\mbox{Var}(\\boldsymbol{\\hat{\\theta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})] [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})]^{T} \\}\n", "\\\\\n", - "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}]^{T} \\}\n", + "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}]^{T} \\}\n", "\\\\\n", - "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", + "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", "% \\\\\n", - "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\boldsymbol{\\beta}^T\n", + "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\boldsymbol{\\theta}^T\n", "\\\\\n", - "& = & \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\, \\, \\, = \\, \\, \\, \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1},\n", "\\end{eqnarray*}\n", "$$" @@ -322,21 +322,21 @@ }, { "cell_type": "markdown", - "id": "3285f57f", + "id": "5d8dd6bc", "metadata": { "editable": true }, "source": [ "where we have used that $\\mathbb{E} (\\mathbf{Y} \\mathbf{Y}^{T}) =\n", - "\\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} +\n", - "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\beta}) = \\sigma^2\n", + "\\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} +\n", + "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\theta}) = \\sigma^2\n", "\\, (\\mathbf{X}^{T} \\mathbf{X})^{-1}$, one obtains an estimate of the\n", "variance of the estimate of the $j$-th regression coefficient:\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", "construct a confidence interval for the estimates.\n", "\n", "In a similar way, we can obtain analytical expressions for say the\n", - "expectation values of the parameters $\\boldsymbol{\\beta}$ and their variance\n", + "expectation values of the parameters $\\boldsymbol{\\theta}$ and their variance\n", "when we employ Ridge regression, allowing us again to define a confidence interval. \n", "\n", "It is rather straightforward to show that" @@ -344,80 +344,80 @@ }, { "cell_type": "markdown", - "id": "925312c4", + "id": "3594078a", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\beta}^{\\mathrm{OLS}}.\n", + "\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\theta}^{\\mathrm{OLS}}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "b9fd6bed", + "id": "43007b7a", "metadata": { "editable": true }, "source": [ "We see clearly that \n", - "$\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\beta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", + "$\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\theta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", "\n", "We can also compute the variance as" ] }, { "cell_type": "markdown", - "id": "d919b998", + "id": "3cd4e7da", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", "$$" ] }, { "cell_type": "markdown", - "id": "a878a50d", + "id": "5816b21e", "metadata": { "editable": true }, "source": [ - "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\beta}$ goes to zero. \n", + "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\theta}$ goes to zero. \n", "\n", "With this, we can compute the difference" ] }, { "cell_type": "markdown", - "id": "ccbbea14", + "id": "b58b8607", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\beta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\theta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "74775fd2", + "id": "c843db7f", "metadata": { "editable": true }, "source": [ "The difference is non-negative definite since each component of the\n", "matrix product is non-negative definite. \n", - "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." + "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\theta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." ] }, { "cell_type": "markdown", - "id": "032b9317", + "id": "e81b1c39", "metadata": { "editable": true }, @@ -431,28 +431,28 @@ "$\\sigma^2$.\n", "\n", "We found above that the outputs $\\boldsymbol{y}$ have a mean value given by\n", - "$\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$ and variance $\\sigma^2$. Since the entries to\n", + "$\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$ and variance $\\sigma^2$. Since the entries to\n", "the design matrix are not stochastic variables, we can assume that the\n", "probability distribution of our targets is also a normal distribution\n", - "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$. This means that a\n", + "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$. This means that a\n", "single output $y_i$ is given by the Gaussian distribution" ] }, { "cell_type": "markdown", - "id": "1c61251d", + "id": "ea9719e3", "metadata": { "editable": true }, "source": [ "$$\n", - "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "5545f5c8", + "id": "c28b598f", "metadata": { "editable": true }, @@ -465,43 +465,43 @@ }, { "cell_type": "markdown", - "id": "470b75b2", + "id": "88063ed7", "metadata": { "editable": true }, "source": [ "$$\n", - "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]},\n", + "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]},\n", "$$" ] }, { "cell_type": "markdown", - "id": "908e209b", + "id": "a34e4c28", "metadata": { "editable": true }, "source": [ - "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\beta}$.\n", + "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\theta}$.\n", "\n", "Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event $\\boldsymbol{y}$ as the product of the single events, that is we have" ] }, { "cell_type": "markdown", - "id": "4fdbf729", + "id": "1fc7ae0a", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta}).\n", + "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta}).\n", "$$" ] }, { "cell_type": "markdown", - "id": "ac5dc52a", + "id": "30d5a1be", "metadata": { "editable": true }, @@ -512,7 +512,7 @@ }, { "cell_type": "markdown", - "id": "dc35e3e1", + "id": "1d29fd29", "metadata": { "editable": true }, @@ -524,7 +524,7 @@ }, { "cell_type": "markdown", - "id": "f0b84ac4", + "id": "289a116c", "metadata": { "editable": true }, @@ -535,29 +535,29 @@ }, { "cell_type": "markdown", - "id": "90eb85ae", + "id": "d1cfbb56", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{D}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "p(\\boldsymbol{D}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "5e70b7b1", + "id": "e089fc0e", "metadata": { "editable": true }, "source": [ - "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\beta}$." + "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\theta}$." ] }, { "cell_type": "markdown", - "id": "dcf0e487", + "id": "32cf9944", "metadata": { "editable": true }, @@ -571,7 +571,7 @@ "data is the most probable. \n", "\n", "We will assume here that our events are given by the above Gaussian\n", - "distribution and we will determine the optimal parameters $\\beta$ by\n", + "distribution and we will determine the optimal parameters $\\theta$ by\n", "maximizing the above PDF. However, computing the derivatives of a\n", "product function is cumbersome and can easily lead to overflow and/or\n", "underflowproblems, with potentials for loss of numerical precision.\n", @@ -588,7 +588,7 @@ }, { "cell_type": "markdown", - "id": "c14bbd71", + "id": "ee8a9544", "metadata": { "editable": true }, @@ -600,19 +600,19 @@ }, { "cell_type": "markdown", - "id": "13fcb784", + "id": "4bfeb203", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})},\n", + "C(\\boldsymbol{\\theta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})},\n", "$$" ] }, { "cell_type": "markdown", - "id": "8f685501", + "id": "256143f9", "metadata": { "editable": true }, @@ -622,63 +622,63 @@ }, { "cell_type": "markdown", - "id": "1b83ae5f", + "id": "60d75bb1", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})\\vert\\vert_2^2}{2\\sigma^2}.\n", + "C(\\boldsymbol{\\theta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta})\\vert\\vert_2^2}{2\\sigma^2}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "c44b7952", + "id": "7b111014", "metadata": { "editable": true }, "source": [ - "Taking the derivative of the *new* cost function with respect to the parameters $\\beta$ we recognize our familiar OLS equation, namely" + "Taking the derivative of the *new* cost function with respect to the parameters $\\theta$ we recognize our familiar OLS equation, namely" ] }, { "cell_type": "markdown", - "id": "7848c18a", + "id": "0f42a0b5", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right) =0,\n", + "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta}\\right) =0,\n", "$$" ] }, { "cell_type": "markdown", - "id": "bd67b48f", + "id": "39a9276b", "metadata": { "editable": true }, "source": [ - "which leads to the well-known OLS equation for the optimal paramters $\\beta$" + "which leads to the well-known OLS equation for the optimal paramters $\\theta$" ] }, { "cell_type": "markdown", - "id": "8a8665d9", + "id": "84c63927", "metadata": { "editable": true }, "source": [ "$$\n", - "\\hat{\\boldsymbol{\\beta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", + "\\hat{\\boldsymbol{\\theta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", "$$" ] }, { "cell_type": "markdown", - "id": "718fe69c", + "id": "c087ef22", "metadata": { "editable": true }, @@ -688,7 +688,7 @@ }, { "cell_type": "markdown", - "id": "f837afb3", + "id": "79503987", "metadata": { "editable": true }, @@ -707,7 +707,7 @@ }, { "cell_type": "markdown", - "id": "82ed1dd7", + "id": "c165025b", "metadata": { "editable": true }, @@ -735,7 +735,7 @@ }, { "cell_type": "markdown", - "id": "06046f8f", + "id": "efb63405", "metadata": { "editable": true }, @@ -761,7 +761,7 @@ }, { "cell_type": "markdown", - "id": "7639ed0d", + "id": "88d4bfd8", "metadata": { "editable": true }, @@ -778,7 +778,7 @@ }, { "cell_type": "markdown", - "id": "dc080604", + "id": "6a0795e0", "metadata": { "editable": true }, @@ -798,7 +798,7 @@ }, { "cell_type": "markdown", - "id": "11aa170a", + "id": "d815fdf3", "metadata": { "editable": true }, @@ -827,7 +827,7 @@ }, { "cell_type": "markdown", - "id": "a56444a1", + "id": "a5b26ed1", "metadata": { "editable": true }, @@ -852,7 +852,7 @@ }, { "cell_type": "markdown", - "id": "020c8c0c", + "id": "e5783b81", "metadata": { "editable": true }, @@ -872,7 +872,7 @@ }, { "cell_type": "markdown", - "id": "f707010c", + "id": "9bdfff4f", "metadata": { "editable": true }, @@ -884,7 +884,7 @@ }, { "cell_type": "markdown", - "id": "062d8d7c", + "id": "bd1e4a83", "metadata": { "editable": true }, @@ -894,7 +894,7 @@ }, { "cell_type": "markdown", - "id": "adf6bd14", + "id": "66d14a83", "metadata": { "editable": true }, @@ -909,7 +909,7 @@ }, { "cell_type": "markdown", - "id": "aaa71fca", + "id": "ac38d462", "metadata": { "editable": true }, @@ -922,7 +922,7 @@ }, { "cell_type": "markdown", - "id": "83af27c8", + "id": "65488a60", "metadata": { "editable": true }, @@ -935,7 +935,7 @@ }, { "cell_type": "markdown", - "id": "e22a05bb", + "id": "9c8686d8", "metadata": { "editable": true }, @@ -947,7 +947,7 @@ }, { "cell_type": "markdown", - "id": "2577f9d3", + "id": "585dbaff", "metadata": { "editable": true }, @@ -960,7 +960,7 @@ }, { "cell_type": "markdown", - "id": "244731e1", + "id": "a8093a3e", "metadata": { "editable": true }, @@ -971,7 +971,7 @@ }, { "cell_type": "markdown", - "id": "37657b97", + "id": "ec844baa", "metadata": { "editable": true }, @@ -985,7 +985,7 @@ }, { "cell_type": "markdown", - "id": "e2c5221a", + "id": "7f2b563b", "metadata": { "editable": true }, @@ -995,7 +995,7 @@ }, { "cell_type": "markdown", - "id": "68bcc41e", + "id": "5a75bbe4", "metadata": { "editable": true }, @@ -1009,7 +1009,7 @@ }, { "cell_type": "markdown", - "id": "bf00fd90", + "id": "8d00a9ff", "metadata": { "editable": true }, @@ -1022,7 +1022,7 @@ }, { "cell_type": "markdown", - "id": "27c016a4", + "id": "cae54a40", "metadata": { "editable": true }, @@ -1035,7 +1035,7 @@ }, { "cell_type": "markdown", - "id": "a9ab9ec0", + "id": "ef87a15a", "metadata": { "editable": true }, @@ -1045,7 +1045,7 @@ }, { "cell_type": "markdown", - "id": "016819e4", + "id": "5bfaef5d", "metadata": { "editable": true }, @@ -1058,7 +1058,7 @@ }, { "cell_type": "markdown", - "id": "e5d0c294", + "id": "0aab5f59", "metadata": { "editable": true }, @@ -1068,7 +1068,7 @@ }, { "cell_type": "markdown", - "id": "c07c8ef9", + "id": "71269afb", "metadata": { "editable": true }, @@ -1081,7 +1081,7 @@ }, { "cell_type": "markdown", - "id": "ca560de7", + "id": "4d809fab", "metadata": { "editable": true }, @@ -1093,7 +1093,7 @@ }, { "cell_type": "markdown", - "id": "0b3fb5f5", + "id": "734f7c5e", "metadata": { "editable": true }, @@ -1112,7 +1112,7 @@ }, { "cell_type": "markdown", - "id": "0ee945bc", + "id": "f4ff2861", "metadata": { "editable": true }, @@ -1125,7 +1125,7 @@ }, { "cell_type": "markdown", - "id": "5e0d7232", + "id": "fd66a47c", "metadata": { "editable": true }, @@ -1137,7 +1137,7 @@ }, { "cell_type": "markdown", - "id": "4f0c89e3", + "id": "51e5434b", "metadata": { "editable": true }, @@ -1150,7 +1150,7 @@ }, { "cell_type": "markdown", - "id": "c2a810e6", + "id": "9129f7a5", "metadata": { "editable": true }, @@ -1170,7 +1170,7 @@ }, { "cell_type": "markdown", - "id": "58260444", + "id": "e5fa7f9c", "metadata": { "editable": true }, @@ -1179,13 +1179,13 @@ "\n", "Confidence intervals are used in statistics and represent a type of estimate\n", "computed from the observed data. This gives a range of values for an\n", - "unknown parameter such as the parameters $\\boldsymbol{\\beta}$ from linear regression.\n", + "unknown parameter such as the parameters $\\boldsymbol{\\theta}$ from linear regression.\n", "\n", - "With the OLS expressions for the parameters $\\boldsymbol{\\beta}$ we found \n", - "$\\mathbb{E}(\\boldsymbol{\\beta}) = \\boldsymbol{\\beta}$, which means that the estimator of the regression parameters is unbiased.\n", + "With the OLS expressions for the parameters $\\boldsymbol{\\theta}$ we found \n", + "$\\mathbb{E}(\\boldsymbol{\\theta}) = \\boldsymbol{\\theta}$, which means that the estimator of the regression parameters is unbiased.\n", "\n", "In the exercises this week we show that the variance of the estimate of the $j$-th regression coefficient is\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", "\n", "This quantity can be used to\n", "construct a confidence interval for the estimates." @@ -1193,34 +1193,34 @@ }, { "cell_type": "markdown", - "id": "24ddf9fe", + "id": "f5bbafe6", "metadata": { "editable": true }, "source": [ "## Standard Approach based on the Normal Distribution\n", "\n", - "We will assume that the parameters $\\beta$ follow a normal\n", + "We will assume that the parameters $\\theta$ follow a normal\n", "distribution. We can then define the confidence interval. Here we will be using as\n", - "shorthands $\\mu_{\\beta}$ for the above mean value and $\\sigma_{\\beta}$\n", + "shorthands $\\mu_{\\theta}$ for the above mean value and $\\sigma_{\\theta}$\n", "for the standard deviation. We have then a confidence interval" ] }, { "cell_type": "markdown", - "id": "459e4649", + "id": "0cc93a0c", "metadata": { "editable": true }, "source": [ "$$\n", - "\\left(\\mu_{\\beta}\\pm \\frac{z\\sigma_{\\beta}}{\\sqrt{n}}\\right),\n", + "\\left(\\mu_{\\theta}\\pm \\frac{z\\sigma_{\\theta}}{\\sqrt{n}}\\right),\n", "$$" ] }, { "cell_type": "markdown", - "id": "8924d816", + "id": "8dd4f5e0", "metadata": { "editable": true }, @@ -1240,18 +1240,18 @@ }, { "cell_type": "markdown", - "id": "82ec4b91", + "id": "f51c546c", "metadata": { "editable": true }, "source": [ "## Resampling methods: Bootstrap background\n", "\n", - "Since $\\widehat{\\beta} = \\widehat{\\beta}(\\boldsymbol{X})$ is a function of random variables,\n", - "$\\widehat{\\beta}$ itself must be a random variable. Thus it has\n", + "Since $\\widehat{\\theta} = \\widehat{\\theta}(\\boldsymbol{X})$ is a function of random variables,\n", + "$\\widehat{\\theta}$ itself must be a random variable. Thus it has\n", "a pdf, call this function $p(\\boldsymbol{t})$. The aim of the bootstrap is to\n", "estimate $p(\\boldsymbol{t})$ by the relative frequency of\n", - "$\\widehat{\\beta}$. You can think of this as using a histogram\n", + "$\\widehat{\\theta}$. You can think of this as using a histogram\n", "in the place of $p(\\boldsymbol{t})$. If the relative frequency closely\n", "resembles $p(\\vec{t})$, then using numerics, it is straight forward to\n", "estimate all the interesting parameters of $p(\\boldsymbol{t})$ using point\n", @@ -1260,31 +1260,31 @@ }, { "cell_type": "markdown", - "id": "6e194c67", + "id": "e44fcf6d", "metadata": { "editable": true }, "source": [ "## Resampling methods: More Bootstrap background\n", "\n", - "In the case that $\\widehat{\\beta}$ has\n", + "In the case that $\\widehat{\\theta}$ has\n", "more than one component, and the components are independent, we use the\n", "same estimator on each component separately. If the probability\n", "density function of $X_i$, $p(x)$, had been known, then it would have\n", "been straightforward to do this by: \n", "1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \\cdots, X_n^*)$. \n", "\n", - "2. Then using these numbers, we could compute a replica of $\\widehat{\\beta}$ called $\\widehat{\\beta}^*$. \n", + "2. Then using these numbers, we could compute a replica of $\\widehat{\\theta}$ called $\\widehat{\\theta}^*$. \n", "\n", "By repeated use of the above two points, many\n", - "estimates of $\\widehat{\\beta}$ can be obtained. The\n", - "idea is to use the relative frequency of $\\widehat{\\beta}^*$\n", + "estimates of $\\widehat{\\theta}$ can be obtained. The\n", + "idea is to use the relative frequency of $\\widehat{\\theta}^*$\n", "(think of a histogram) as an estimate of $p(\\boldsymbol{t})$." ] }, { "cell_type": "markdown", - "id": "af32bd4f", + "id": "3bd69373", "metadata": { "editable": true }, @@ -1305,7 +1305,7 @@ }, { "cell_type": "markdown", - "id": "66865206", + "id": "e7f867d9", "metadata": { "editable": true }, @@ -1318,24 +1318,24 @@ "\n", "2. Define a vector $\\boldsymbol{x}^*$ containing the values which were drawn from $\\boldsymbol{x}$. \n", "\n", - "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\beta}^*$ by evaluating $\\widehat \\beta$ under the observations $\\boldsymbol{x}^*$. \n", + "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\theta}^*$ by evaluating $\\widehat \\theta$ under the observations $\\boldsymbol{x}^*$. \n", "\n", "4. Repeat this process $k$ times. \n", "\n", "When you are done, you can draw a histogram of the relative frequency\n", - "of $\\widehat \\beta^*$. This is your estimate of the probability\n", + "of $\\widehat \\theta^*$. This is your estimate of the probability\n", "distribution $p(t)$. Using this probability distribution you can\n", "estimate any statistics thereof. In principle you never draw the\n", - "histogram of the relative frequency of $\\widehat{\\beta}^*$. Instead\n", + "histogram of the relative frequency of $\\widehat{\\theta}^*$. Instead\n", "you use the estimators corresponding to the statistic of interest. For\n", "example, if you are interested in estimating the variance of $\\widehat\n", - "\\beta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", - "$\\widehat \\beta^*$." + "\\theta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", + "$\\widehat \\theta^*$." ] }, { "cell_type": "markdown", - "id": "c9c40149", + "id": "5c2c3909", "metadata": { "editable": true }, @@ -1359,7 +1359,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "e6ac04cd", + "id": "a32faf6a", "metadata": { "collapsed": false, "editable": true @@ -1398,7 +1398,7 @@ }, { "cell_type": "markdown", - "id": "b490ce43", + "id": "bc95505d", "metadata": { "editable": true }, @@ -1408,7 +1408,7 @@ }, { "cell_type": "markdown", - "id": "ce2afb67", + "id": "355f62af", "metadata": { "editable": true }, @@ -1419,7 +1419,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "b194e79e", + "id": "39fd1fa8", "metadata": { "collapsed": false, "editable": true @@ -1439,7 +1439,7 @@ }, { "cell_type": "markdown", - "id": "e557245e", + "id": "237b4db3", "metadata": { "editable": true }, @@ -1457,7 +1457,7 @@ }, { "cell_type": "markdown", - "id": "c30637d8", + "id": "0c13e5b0", "metadata": { "editable": true }, @@ -1469,7 +1469,7 @@ }, { "cell_type": "markdown", - "id": "e5c68caa", + "id": "a086aa7e", "metadata": { "editable": true }, @@ -1478,27 +1478,27 @@ "\n", "In our derivation of the ordinary least squares method we defined then\n", "an approximation to the function $f$ in terms of the parameters\n", - "$\\boldsymbol{\\beta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", - "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\beta}$. \n", + "$\\boldsymbol{\\theta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", + "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\theta}$. \n", "\n", - "Thereafter we found the parameters $\\boldsymbol{\\beta}$ by optimizing the means squared error via the so-called cost function" + "Thereafter we found the parameters $\\boldsymbol{\\theta}$ by optimizing the means squared error via the so-called cost function" ] }, { "cell_type": "markdown", - "id": "d0c891d5", + "id": "c3837d89", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{X},\\boldsymbol{\\beta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", + "C(\\boldsymbol{X},\\boldsymbol{\\theta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", "$$" ] }, { "cell_type": "markdown", - "id": "eb41d38c", + "id": "7c7cd0a7", "metadata": { "editable": true }, @@ -1508,7 +1508,7 @@ }, { "cell_type": "markdown", - "id": "8e74cdc8", + "id": "b9db0cd5", "metadata": { "editable": true }, @@ -1520,7 +1520,7 @@ }, { "cell_type": "markdown", - "id": "b135304d", + "id": "00482b2d", "metadata": { "editable": true }, @@ -1537,7 +1537,7 @@ }, { "cell_type": "markdown", - "id": "5fae4e71", + "id": "2e7c7291", "metadata": { "editable": true }, @@ -1549,7 +1549,7 @@ }, { "cell_type": "markdown", - "id": "4e89ca72", + "id": "9a8d29b5", "metadata": { "editable": true }, @@ -1559,7 +1559,7 @@ }, { "cell_type": "markdown", - "id": "a2edd75d", + "id": "9a8e2084", "metadata": { "editable": true }, @@ -1571,7 +1571,7 @@ }, { "cell_type": "markdown", - "id": "0dfbc890", + "id": "4a191c5c", "metadata": { "editable": true }, @@ -1581,7 +1581,7 @@ }, { "cell_type": "markdown", - "id": "4168fa86", + "id": "c37713f8", "metadata": { "editable": true }, @@ -1593,7 +1593,7 @@ }, { "cell_type": "markdown", - "id": "defe083c", + "id": "82ca4a7b", "metadata": { "editable": true }, @@ -1603,7 +1603,7 @@ }, { "cell_type": "markdown", - "id": "ed067301", + "id": "2765a841", "metadata": { "editable": true }, @@ -1619,7 +1619,7 @@ }, { "cell_type": "markdown", - "id": "815dc960", + "id": "321964c1", "metadata": { "editable": true }, @@ -1630,7 +1630,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "1ca319f9", + "id": "5942226f", "metadata": { "collapsed": false, "editable": true @@ -1695,7 +1695,7 @@ }, { "cell_type": "markdown", - "id": "7c9d4971", + "id": "cf2af19d", "metadata": { "editable": true }, @@ -1706,7 +1706,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "3b1f1345", + "id": "7b371a0c", "metadata": { "collapsed": false, "editable": true @@ -1763,7 +1763,7 @@ }, { "cell_type": "markdown", - "id": "a1dfc4a3", + "id": "6f583fcd", "metadata": { "editable": true }, @@ -1801,7 +1801,7 @@ }, { "cell_type": "markdown", - "id": "1b0835bb", + "id": "37766c51", "metadata": { "editable": true }, @@ -1828,7 +1828,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "3c6f8392", + "id": "0eac5ca9", "metadata": { "collapsed": false, "editable": true @@ -1890,7 +1890,7 @@ }, { "cell_type": "markdown", - "id": "195d6e77", + "id": "d698096e", "metadata": { "editable": true }, @@ -1915,7 +1915,7 @@ }, { "cell_type": "markdown", - "id": "1458a723", + "id": "3e37a4d2", "metadata": { "editable": true }, @@ -1943,7 +1943,7 @@ }, { "cell_type": "markdown", - "id": "d5a79ca2", + "id": "e7e43cf9", "metadata": { "editable": true }, @@ -1956,7 +1956,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "65b5aaec", + "id": "7aa6a568", "metadata": { "collapsed": false, "editable": true @@ -2056,7 +2056,7 @@ }, { "cell_type": "markdown", - "id": "d57db05f", + "id": "9a90aec7", "metadata": { "editable": true }, @@ -2067,7 +2067,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "bf4fcde2", + "id": "93b3af24", "metadata": { "collapsed": false, "editable": true @@ -2156,7 +2156,7 @@ }, { "cell_type": "markdown", - "id": "ea3509c6", + "id": "90657c6d", "metadata": { "editable": true }, @@ -2166,7 +2166,7 @@ }, { "cell_type": "markdown", - "id": "95edcb8b", + "id": "a88b86e7", "metadata": { "editable": true }, @@ -2179,7 +2179,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "efed8b4e", + "id": "89f45188", "metadata": { "collapsed": false, "editable": true @@ -2257,7 +2257,7 @@ }, { "cell_type": "markdown", - "id": "8dadd613", + "id": "75496517", "metadata": { "editable": true }, diff --git a/doc/LectureNotes/_build/html/searchindex.js b/doc/LectureNotes/_build/html/searchindex.js index fa5659908..b5568cb04 100644 --- a/doc/LectureNotes/_build/html/searchindex.js +++ b/doc/LectureNotes/_build/html/searchindex.js @@ -1 +1 @@ -Search.setIndex({"alltitles": {"1a)": [[18, "a"]], "3a)": [[18, "id1"]], "3b)": [[18, "b"]], "4a)": [[18, "id2"]], "4b)": [[18, "id3"]], "A Classification Tree": [[9, "a-classification-tree"]], "A Frequentist approach to data analysis": [[0, "a-frequentist-approach-to-data-analysis"], [28, "a-frequentist-approach-to-data-analysis"]], "A better approach": [[8, "a-better-approach"]], "A first summary": [[28, "a-first-summary"]], "A new Cost Function": [[32, "a-new-cost-function"]], "A quick Reminder on Lagrangian Multipliers": [[8, "a-quick-reminder-on-lagrangian-multipliers"]], "A simple example": [[4, "a-simple-example"]], "A soft classifier": [[8, "a-soft-classifier"]], "A top-down perspective on Neural networks": [[1, "a-top-down-perspective-on-neural-networks"]], "A way to Read the Bias-Variance Tradeoff": [[32, "a-way-to-read-the-bias-variance-tradeoff"]], "ADAM algorithm, taken from Goodfellow et al": [[31, "adam-algorithm-taken-from-goodfellow-et-al"]], "ADAM optimizer": [[13, "adam-optimizer"], [31, "id2"]], "Accuracy": [[31, "accuracy"]], "Activation functions": [[12, "activation-functions"]], "AdaGrad Properties": [[31, "adagrad-properties"]], "AdaGrad Update Rule Derivation": [[31, "adagrad-update-rule-derivation"]], "AdaGrad algorithm, taken from Goodfellow et al": [[31, "adagrad-algorithm-taken-from-goodfellow-et-al"]], "Adam Optimizer": [[31, "adam-optimizer"]], "Adam vs. AdaGrad and RMSProp": [[31, "adam-vs-adagrad-and-rmsprop"]], "Adam: Bias Correction": [[31, "adam-bias-correction"]], "Adam: Exponential Moving Averages (Moments)": [[31, "adam-exponential-moving-averages-moments"]], "Adam: Update Rule Derivation": [[31, "adam-update-rule-derivation"]], "Adaptive boosting: AdaBoost, Basic Algorithm": [[10, "adaptive-boosting-adaboost-basic-algorithm"]], "Adaptivity Across Dimensions": [[31, "adaptivity-across-dimensions"]], "Adding error analysis and training set up": [[28, "adding-error-analysis-and-training-set-up"], [29, "adding-error-analysis-and-training-set-up"]], "Adjust hyperparameters": [[1, "adjust-hyperparameters"]], "Algorithms and codes for Adagrad, RMSprop and Adam": [[31, "algorithms-and-codes-for-adagrad-rmsprop-and-adam"]], "Algorithms for Setting up Decision Trees": [[9, "algorithms-for-setting-up-decision-trees"]], "An Overview of Ensemble Methods": [[10, "an-overview-of-ensemble-methods"]], "An extrapolation example": [[4, "an-extrapolation-example"]], "An optimization/minimization problem": [[28, "an-optimization-minimization-problem"]], "And finally \\boldsymbol{X}\\boldsymbol{X}^T": [[29, "and-finally-boldsymbol-x-boldsymbol-x-t"]], "And finally ADAM": [[31, "and-finally-adam"]], "And what about using neural networks?": [[28, "and-what-about-using-neural-networks"]], "Another Example from Scikit-Learn\u2019s Repository": [[32, "another-example-from-scikit-learn-s-repository"]], "Another Example, now with a polynomial fit": [[30, "another-example-now-with-a-polynomial-fit"]], "Another example, the moons again": [[9, "another-example-the-moons-again"]], "Applied Data Analysis and Machine Learning": [[21, null]], "Assumptions made": [[32, "assumptions-made"]], "Autocorrelation function": [[25, "autocorrelation-function"]], "Automatic differentiation": [[13, "automatic-differentiation"]], "Back to Ridge and LASSO Regression": [[29, "back-to-ridge-and-lasso-regression"], [30, "back-to-ridge-and-lasso-regression"]], "Back to the Cancer Data": [[11, "back-to-the-cancer-data"]], "Background literature": [[23, "background-literature"]], "Bagging": [[10, "bagging"]], "Bagging Examples": [[10, "bagging-examples"]], "Basic Matrix Features": [[22, "basic-matrix-features"]], "Basic ideas of the Principal Component Analysis (PCA)": [[11, null]], "Basic math of the SVD": [[5, "basic-math-of-the-svd"], [29, "basic-math-of-the-svd"], [30, "basic-math-of-the-svd"]], "Basics": [[7, "basics"]], "Basics of a tree": [[9, "basics-of-a-tree"]], "Batch Normalization": [[1, "batch-normalization"]], "Batches and mini-batches": [[31, "batches-and-mini-batches"]], "Bayes\u2019 Theorem and Ridge and Lasso Regression": [[5, "bayes-theorem-and-ridge-and-lasso-regression"]], "Boosting, a Bird\u2019s Eye View": [[10, "boosting-a-bird-s-eye-view"]], "Bootstrap": [[6, "bootstrap"]], "Bringing it together, first back propagation equation": [[12, "bringing-it-together-first-back-propagation-equation"]], "Building a Feed Forward Neural Network": [[1, null]], "Building a tree, regression": [[9, "building-a-tree-regression"]], "Building neural networks in Tensorflow and Keras": [[1, "building-neural-networks-in-tensorflow-and-keras"]], "But none of these can compete with Newton\u2019s method": [[31, "but-none-of-these-can-compete-with-newton-s-method"]], "CNNs in more detail, building convolutional neural networks in Tensorflow and Keras": [[3, "cnns-in-more-detail-building-convolutional-neural-networks-in-tensorflow-and-keras"]], "Cancer Data again now with Decision Trees and other Methods": [[9, "cancer-data-again-now-with-decision-trees-and-other-methods"]], "Challenge: Choosing a Fixed Learning Rate": [[31, "challenge-choosing-a-fixed-learning-rate"]], "Choose cost function and optimizer": [[1, "choose-cost-function-and-optimizer"]], "Classical PCA Theorem": [[11, "classical-pca-theorem"]], "Clustering and Unsupervised Learning": [[14, null]], "Code Example for Cross-validation and k-fold Cross-validation": [[32, "code-example-for-cross-validation-and-k-fold-cross-validation"]], "Code example for the Bootstrap method": [[32, "code-example-for-the-bootstrap-method"]], "Code for SVD and Inversion of Matrices": [[5, "code-for-svd-and-inversion-of-matrices"]], "Code with a Number of Minibatches which varies": [[31, "code-with-a-number-of-minibatches-which-varies"]], "Codes and Approaches": [[14, "codes-and-approaches"]], "Codes for the SVD": [[5, "codes-for-the-svd"], [29, "codes-for-the-svd"], [30, "codes-for-the-svd"]], "Coding Setup and Linear Regression": [[15, "coding-setup-and-linear-regression"]], "Collect and pre-process data": [[1, "collect-and-pre-process-data"]], "Communication channels": [[28, "communication-channels"]], "Compare Bagging on Trees with Random Forests": [[10, "compare-bagging-on-trees-with-random-forests"]], "Comparing with a numerical scheme": [[2, "comparing-with-a-numerical-scheme"]], "Comparison with OLS": [[30, "comparison-with-ols"]], "Computation of gradients": [[31, "computation-of-gradients"]], "Computing the Gini index": [[9, "computing-the-gini-index"]], "Conditions on convex functions": [[30, "conditions-on-convex-functions"]], "Confidence Intervals": [[32, "confidence-intervals"]], "Conjugate gradient method": [[13, "conjugate-gradient-method"]], "Convergence rates": [[31, "convergence-rates"]], "Convex function": [[30, "convex-function"]], "Convex functions": [[13, "convex-functions"], [30, "convex-functions"]], "Convolution Examples: Polynomial multiplication": [[3, "convolution-examples-polynomial-multiplication"]], "Convolution Examples: Principle of Superposition and Periodic Forces (Fourier Transforms)": [[3, "convolution-examples-principle-of-superposition-and-periodic-forces-fourier-transforms"]], "Convolutional Neural Network": [[12, "convolutional-neural-network"]], "Convolutional Neural Networks": [[3, null]], "Correlation Function and Design/Feature Matrix": [[29, "correlation-function-and-design-feature-matrix"]], "Correlation Matrix": [[11, "correlation-matrix"], [29, "correlation-matrix"]], "Correlation Matrix with Pandas": [[29, "correlation-matrix-with-pandas"]], "Course Format": [[28, "course-format"]], "Course setting": [[24, null]], "Covariance Matrix Examples": [[29, "covariance-matrix-examples"]], "Covariance and Correlation Matrix": [[29, "covariance-and-correlation-matrix"]], "Cross-validation": [[6, "cross-validation"]], "Cross-validation in brief": [[32, "cross-validation-in-brief"]], "Deadlines for projects (tentative)": [[28, "deadlines-for-projects-tentative"]], "Decision trees, overarching aims": [[9, null]], "Deep Neural Networks": [[31, "deep-neural-networks"]], "Deep learning methods": [[28, "deep-learning-methods"]], "Define model and architecture": [[1, "define-model-and-architecture"]], "Defining the cost function": [[1, "defining-the-cost-function"]], "Definitions": [[19, "definitions"]], "Deliverables": [[15, "deliverables"], [16, "deliverables"], [19, "deliverables"], [20, "deliverables"]], "Derivation of the AdaGrad Algorithm": [[31, "derivation-of-the-adagrad-algorithm"]], "Derivatives and the chain rule": [[12, "derivatives-and-the-chain-rule"]], "Derivatives, example 1": [[29, "derivatives-example-1"]], "Deriving OLS from a probability distribution": [[5, "deriving-ols-from-a-probability-distribution"], [32, "deriving-ols-from-a-probability-distribution"]], "Deriving and Implementing Ordinary Least Squares": [[16, "deriving-and-implementing-ordinary-least-squares"]], "Deriving and Implementing Ridge Regression": [[17, "deriving-and-implementing-ridge-regression"]], "Deriving the Lasso Regression Equations": [[29, "deriving-the-lasso-regression-equations"], [30, "deriving-the-lasso-regression-equations"], [30, "id6"]], "Deriving the Ridge Regression Equations": [[29, "deriving-the-ridge-regression-equations"], [30, "deriving-the-ridge-regression-equations"], [30, "id3"]], "Deriving the back propagation code for a multilayer perceptron model": [[12, "deriving-the-back-propagation-code-for-a-multilayer-perceptron-model"]], "Developing a code for doing neural networks with back propagation": [[1, "developing-a-code-for-doing-neural-networks-with-back-propagation"]], "Diagonalize the sample covariance matrix to obtain the principal components": [[11, "diagonalize-the-sample-covariance-matrix-to-obtain-the-principal-components"]], "Different kernels and Mercer\u2019s theorem": [[8, "different-kernels-and-mercer-s-theorem"]], "Disadvantages": [[9, "disadvantages"]], "Discriminative Modeling": [[28, "discriminative-modeling"]], "Domains and probabilities": [[25, "domains-and-probabilities"]], "Dropout": [[1, "dropout"]], "Economy-size SVD": [[29, "economy-size-svd"], [30, "economy-size-svd"]], "Elements of Probability Theory and Statistical Data Analysis": [[25, null]], "Empirical Evidence: Convergence Time and Memory in Practice": [[31, "empirical-evidence-convergence-time-and-memory-in-practice"]], "Ensemble Methods: From a Single Tree to Many Trees and Extreme Boosting, Meet the Jungle of Methods": [[10, null]], "Entropy and the ID3 algorithm": [[9, "entropy-and-the-id3-algorithm"]], "Essential elements of ML": [[28, "essential-elements-of-ml"]], "Evaluate model performance on test data": [[1, "evaluate-model-performance-on-test-data"]], "Example 2": [[29, "example-2"]], "Example 3": [[29, "example-3"]], "Example 4": [[29, "example-4"]], "Example Matrix": [[29, "example-matrix"], [30, "example-matrix"]], "Example code for Bias-Variance tradeoff": [[32, "example-code-for-bias-variance-tradeoff"]], "Example of discriminative modeling, taken from Generative Deep Learning by David Foster": [[28, "example-of-discriminative-modeling-taken-from-generative-deep-learning-by-david-foster"]], "Example of generative modeling, taken from Generative Deep Learning by David Foster": [[28, "example-of-generative-modeling-taken-from-generative-deep-learning-by-david-foster"]], "Example of own Standard scaling": [[29, "example-of-own-standard-scaling"]], "Example relevant for the exercises": [[29, "example-relevant-for-the-exercises"]], "Example: Exponential decay": [[2, "example-exponential-decay"]], "Example: Population growth": [[2, "example-population-growth"]], "Example: The diffusion equation": [[2, "example-the-diffusion-equation"]], "Example: binary classification problem": [[1, "example-binary-classification-problem"]], "Examples": [[28, "examples"]], "Examples of likelihood functions used in logistic regression and neural networks": [[7, "examples-of-likelihood-functions-used-in-logistic-regression-and-neural-networks"]], "Exercise 1 - Choice of model and degrees of freedom": [[17, "exercise-1-choice-of-model-and-degrees-of-freedom"]], "Exercise 1 - Finding the derivative of Matrix-Vector expressions": [[16, "exercise-1-finding-the-derivative-of-matrix-vector-expressions"]], "Exercise 1 - Github Setup": [[15, "exercise-1-github-setup"]], "Exercise 1, scale your data": [[18, "exercise-1-scale-your-data"]], "Exercise 1: Creating the report document": [[20, "exercise-1-creating-the-report-document"]], "Exercise 1: Expectation values for ordinary least squares expressions": [[19, "exercise-1-expectation-values-for-ordinary-least-squares-expressions"]], "Exercise 1: Setting up various Python environments": [[0, "exercise-1-setting-up-various-python-environments"]], "Exercise 2 - Deriving the expression for OLS": [[16, "exercise-2-deriving-the-expression-for-ols"]], "Exercise 2 - Deriving the expression for Ridge Regression": [[17, "exercise-2-deriving-the-expression-for-ridge-regression"]], "Exercise 2 - Setting up a Github repository": [[15, "exercise-2-setting-up-a-github-repository"]], "Exercise 2, calculate the gradients": [[18, "exercise-2-calculate-the-gradients"]], "Exercise 2: Adding good figures": [[20, "exercise-2-adding-good-figures"]], "Exercise 2: Expectation values for Ridge regression": [[19, "exercise-2-expectation-values-for-ridge-regression"]], "Exercise 2: making your own data and exploring scikit-learn": [[0, "exercise-2-making-your-own-data-and-exploring-scikit-learn"]], "Exercise 3 - Creating feature matrix and implementing OLS using the analytical expression": [[16, "exercise-3-creating-feature-matrix-and-implementing-ols-using-the-analytical-expression"]], "Exercise 3 - Fitting an OLS model to data": [[15, "exercise-3-fitting-an-ols-model-to-data"]], "Exercise 3 - Scaling data": [[17, "exercise-3-scaling-data"]], "Exercise 3 - Setting up a Python virtual environment": [[15, "exercise-3-setting-up-a-python-virtual-environment"]], "Exercise 3, using the analytical formulae for OLS and Ridge regression to find the optimal paramters \\boldsymbol{\\theta}": [[18, "exercise-3-using-the-analytical-formulae-for-ols-and-ridge-regression-to-find-the-optimal-paramters-boldsymbol-theta"]], "Exercise 3: Deriving the expression for the Bias-Variance Trade-off": [[19, "exercise-3-deriving-the-expression-for-the-bias-variance-trade-off"]], "Exercise 3: Normalizing our data": [[0, "exercise-3-normalizing-our-data"]], "Exercise 3: Writing an abstract and introduction": [[20, "exercise-3-writing-an-abstract-and-introduction"]], "Exercise 4 - Fitting a polynomial": [[16, "exercise-4-fitting-a-polynomial"]], "Exercise 4 - Implementing Ridge Regression": [[17, "exercise-4-implementing-ridge-regression"]], "Exercise 4 - Testing multiple hyperparameters": [[17, "exercise-4-testing-multiple-hyperparameters"]], "Exercise 4 - The train-test split": [[15, "exercise-4-the-train-test-split"]], "Exercise 4, Implementing the simplest form for gradient descent": [[18, "exercise-4-implementing-the-simplest-form-for-gradient-descent"]], "Exercise 4: Adding Ridge Regression": [[0, "exercise-4-adding-ridge-regression"]], "Exercise 4: Computing the Bias and Variance": [[19, "exercise-4-computing-the-bias-and-variance"]], "Exercise 4: Making the code available and presentable": [[20, "exercise-4-making-the-code-available-and-presentable"]], "Exercise 5 - Comparing your code with sklearn": [[16, "exercise-5-comparing-your-code-with-sklearn"]], "Exercise 5, Ridge regression and a new Synthetic Dataset": [[18, "exercise-5-ridge-regression-and-a-new-synthetic-dataset"]], "Exercise 5: Analytical exercises": [[0, "exercise-5-analytical-exercises"]], "Exercise 5: Interpretation of scaling and metrics": [[19, "exercise-5-interpretation-of-scaling-and-metrics"]], "Exercise 5: Referencing": [[20, "exercise-5-referencing"]], "Exercise: Cross-validation as resampling techniques, adding more complexity": [[6, "exercise-cross-validation-as-resampling-techniques-adding-more-complexity"]], "Exercise: Analysis of real data": [[6, "exercise-analysis-of-real-data"]], "Exercise: Bias-variance trade-off and resampling techniques": [[6, "exercise-bias-variance-trade-off-and-resampling-techniques"]], "Exercise: Lasso Regression on the Franke function with resampling": [[6, "exercise-lasso-regression-on-the-franke-function-with-resampling"]], "Exercise: Ordinary Least Square (OLS) on the Franke function": [[6, "exercise-ordinary-least-square-ols-on-the-franke-function"]], "Exercise: Ridge Regression on the Franke function with resampling": [[6, "exercise-ridge-regression-on-the-franke-function-with-resampling"]], "Exercises": [[0, "exercises"]], "Exercises and Projects": [[6, "exercises-and-projects"]], "Exercises week 34": [[15, null]], "Exercises week 35": [[16, null]], "Exercises week 36": [[17, null]], "Exercises week 37": [[18, null]], "Exercises week 38": [[19, null]], "Exercises week 39": [[20, null]], "Expectation value and variance": [[32, "expectation-value-and-variance"]], "Expectation value and variance for \\boldsymbol{\\beta}": [[32, "expectation-value-and-variance-for-boldsymbol-beta"]], "Expectation values": [[25, "expectation-values"]], "Extending to more than one variable": [[30, "extending-to-more-than-one-variable"]], "Extremely useful tools, strongly recommended": [[28, "extremely-useful-tools-strongly-recommended"]], "Feed-forward neural networks": [[12, "feed-forward-neural-networks"]], "Feed-forward pass": [[1, "feed-forward-pass"]], "Final back propagating equation": [[12, "final-back-propagating-equation"]], "Finding the Limit": [[32, "finding-the-limit"]], "Fine-tuning neural network hyperparameters": [[1, "fine-tuning-neural-network-hyperparameters"]], "Fitting an Equation of State for Dense Nuclear Matter": [[0, "fitting-an-equation-of-state-for-dense-nuclear-matter"]], "Fixing the singularity": [[29, "fixing-the-singularity"], [30, "fixing-the-singularity"]], "Format for electronic delivery of report and programs": [[23, "format-for-electronic-delivery-of-report-and-programs"]], "Frequently used scaling functions": [[29, "frequently-used-scaling-functions"], [31, "frequently-used-scaling-functions"]], "From OLS to Ridge and Lasso": [[30, "from-ols-to-ridge-and-lasso"]], "From one to many layers, the universal approximation theorem": [[12, "from-one-to-many-layers-the-universal-approximation-theorem"]], "Functionality in Scikit-Learn": [[29, "functionality-in-scikit-learn"], [31, "functionality-in-scikit-learn"]], "Further Dimensionality Remarks": [[3, "further-dimensionality-remarks"]], "Further properties (important for our analyses later)": [[5, "further-properties-important-for-our-analyses-later"], [29, "further-properties-important-for-our-analyses-later"], [30, "further-properties-important-for-our-analyses-later"]], "Gaussian Elimination": [[22, "gaussian-elimination"]], "General Features": [[9, "general-features"]], "General linear models and linear algebra": [[28, "general-linear-models-and-linear-algebra"]], "Generalizing the fitting procedure as a linear algebra problem": [[28, "generalizing-the-fitting-procedure-as-a-linear-algebra-problem"], [28, "id1"]], "Generative Adversarial Networks": [[4, "generative-adversarial-networks"]], "Generative Models": [[4, "generative-models"]], "Generative Versus Discriminative Modeling": [[28, "generative-versus-discriminative-modeling"]], "Geometric Interpretation and link with Singular Value Decomposition": [[11, "geometric-interpretation-and-link-with-singular-value-decomposition"]], "Getting started with project 1": [[20, "getting-started-with-project-1"]], "Gradient Boosting, Classification Example": [[10, "gradient-boosting-classification-example"]], "Gradient Boosting, Examples of Regression": [[10, "gradient-boosting-examples-of-regression"]], "Gradient Clipping": [[1, "gradient-clipping"]], "Gradient Descent Example": [[30, "id1"], [31, "id1"]], "Gradient boosting: Basics with Steepest Descent/Functional Gradient Descent": [[10, "gradient-boosting-basics-with-steepest-descent-functional-gradient-descent"]], "Gradient descent": [[2, "gradient-descent"]], "Gradient descent and Ridge": [[30, "gradient-descent-and-ridge"], [31, "gradient-descent-and-ridge"]], "Gradient descent and revisiting Ordinary Least Squares from last week": [[31, "gradient-descent-and-revisiting-ordinary-least-squares-from-last-week"]], "Gradient descent example": [[30, "gradient-descent-example"], [31, "gradient-descent-example"]], "Grading": [[26, "grading"], [26, "id2"], [28, "grading"]], "How to take derivatives of Matrix-Vector expressions": [[16, "how-to-take-derivatives-of-matrix-vector-expressions"]], "Hyperplanes and all that": [[8, "hyperplanes-and-all-that"]], "Identifying Terms": [[32, "identifying-terms"]], "Important Matrix and vector handling packages": [[22, "important-matrix-and-vector-handling-packages"]], "Important technicalities: More on Rescaling data": [[29, "important-technicalities-more-on-rescaling-data"]], "Improving gradient descent with momentum": [[31, "improving-gradient-descent-with-momentum"]], "Improving performance": [[1, "improving-performance"]], "In summary": [[26, "in-summary"]], "Including Stochastic Gradient Descent with Autograd": [[13, "including-stochastic-gradient-descent-with-autograd"], [31, "including-stochastic-gradient-descent-with-autograd"]], "Incremental PCA": [[11, "incremental-pca"]], "Independent and Identically Distributed (iid)": [[32, "independent-and-identically-distributed-iid"]], "Installing R, C++, cython or Julia": [[28, "installing-r-c-cython-or-julia"]], "Installing R, C++, cython, Numba etc": [[28, "installing-r-c-cython-numba-etc"]], "Instructor information": [[26, "instructor-information"]], "Interpretations and optimizing our parameters": [[28, "interpretations-and-optimizing-our-parameters"], [28, "id2"], [28, "id3"], [29, "interpretations-and-optimizing-our-parameters"], [29, "id1"], [29, "id2"]], "Interpreting the Ridge results": [[29, "interpreting-the-ridge-results"], [30, "interpreting-the-ridge-results"], [30, "id4"]], "Introducing JAX": [[13, "introducing-jax"]], "Introducing the Covariance and Correlation functions": [[11, "introducing-the-covariance-and-correlation-functions"], [29, "introducing-the-covariance-and-correlation-functions"]], "Introduction": [[0, "introduction"], [6, "introduction"], [21, "introduction"], [22, "introduction"]], "Introduction to numerical projects": [[23, "introduction-to-numerical-projects"]], "Iterative Fitting, Classification and AdaBoost": [[10, "iterative-fitting-classification-and-adaboost"]], "Iterative Fitting, Regression and Squared-error Cost Function": [[10, "iterative-fitting-regression-and-squared-error-cost-function"]], "Kernel PCA": [[11, "kernel-pca"]], "Kernels and non-linearity": [[8, "kernels-and-non-linearity"]], "LU Decomposition, the inverse of a matrix": [[22, "lu-decomposition-the-inverse-of-a-matrix"]], "Lasso Regression": [[30, "lasso-regression"]], "Lasso case": [[30, "lasso-case"]], "Layers": [[1, "layers"]], "Layers used to build CNNs": [[3, "layers-used-to-build-cnns"]], "Learning goals": [[15, "learning-goals"], [16, "learning-goals"], [17, "learning-goals"], [18, "learning-goals"], [19, "learning-goals"], [20, "learning-goals"]], "Learning outcomes": [[21, "learning-outcomes"], [28, "learning-outcomes"]], "Lectures and ComputerLab": [[28, "lectures-and-computerlab"]], "Limitations of supervised learning with deep networks": [[1, "limitations-of-supervised-learning-with-deep-networks"]], "Linear Algebra, Handling of Arrays and more Python Features": [[22, null]], "Linear Regression": [[0, null]], "Linear Regression Problems": [[29, "linear-regression-problems"], [30, "linear-regression-problems"]], "Linear Regression and the SVD": [[30, "linear-regression-and-the-svd"]], "Linear Regression, basic elements": [[0, "linear-regression-basic-elements"]], "Linking Bayes\u2019 Theorem with Ridge and Lasso Regression": [[5, "linking-bayes-theorem-with-ridge-and-lasso-regression"]], "Linking the regression analysis with a statistical interpretation": [[5, "linking-the-regression-analysis-with-a-statistical-interpretation"], [32, "linking-the-regression-analysis-with-a-statistical-interpretation"]], "Linking with the SVD": [[5, "linking-with-the-svd"], [29, "linking-with-the-svd"]], "Links to relevant courses at the University of Oslo": [[27, "links-to-relevant-courses-at-the-university-of-oslo"]], "Logistic Regression": [[7, null], [7, "id1"]], "MNIST and GANs": [[4, "mnist-and-gans"]], "Machine Learning": [[28, "machine-learning"]], "Machine learning": [[21, "machine-learning"]], "Main textbooks": [[28, "main-textbooks"]], "Making a tree": [[9, "making-a-tree"]], "Making your own Bootstrap: Changing the Level of the Decision Tree": [[10, "making-your-own-bootstrap-changing-the-level-of-the-decision-tree"]], "Making your own test-train splitting": [[29, "making-your-own-test-train-splitting"]], "Material for exercises week 35": [[29, "material-for-exercises-week-35"]], "Material for lab sessions sessions Tuesday and Wednesday": [[30, "material-for-lab-sessions-sessions-tuesday-and-wednesday"]], "Material for lecture Monday September 2": [[30, "material-for-lecture-monday-september-2"]], "Material for lecture Monday September 8": [[31, "material-for-lecture-monday-september-8"]], "Material for the lab sessions": [[31, "material-for-the-lab-sessions"], [32, "material-for-the-lab-sessions"]], "Mathematical Interpretation of Ordinary Least Squares": [[5, "mathematical-interpretation-of-ordinary-least-squares"], [29, "mathematical-interpretation-of-ordinary-least-squares"], [30, "mathematical-interpretation-of-ordinary-least-squares"]], "Mathematical optimization of convex functions": [[8, "mathematical-optimization-of-convex-functions"]], "Mathematics of CNNs": [[3, "mathematics-of-cnns"]], "Mathematics of the SVD and implications": [[5, "mathematics-of-the-svd-and-implications"], [29, "mathematics-of-the-svd-and-implications"], [30, "mathematics-of-the-svd-and-implications"]], "Matrices in Python": [[28, "matrices-in-python"]], "Matrix multiplication": [[1, "matrix-multiplication"]], "Matrix-vector notation and activation": [[12, "matrix-vector-notation-and-activation"]], "Maximum Likelihood Estimation (MLE)": [[32, "maximum-likelihood-estimation-mle"]], "Meet the covariance!": [[25, "meet-the-covariance"]], "Meet the Covariance Matrix": [[5, "meet-the-covariance-matrix"], [29, "meet-the-covariance-matrix"]], "Meet the Hessian Matrix": [[29, "meet-the-hessian-matrix"]], "Meet the Pandas": [[28, "meet-the-pandas"]], "Memory Usage and Scalability": [[31, "memory-usage-and-scalability"]], "Memory constraints": [[31, "memory-constraints"]], "Min-Max Scaling": [[29, "min-max-scaling"]], "Momentum based GD": [[13, "momentum-based-gd"], [31, "momentum-based-gd"]], "More complicated Example: The Ising model": [[6, "more-complicated-example-the-ising-model"]], "More examples on bootstrap and cross-validation and errors": [[32, "more-examples-on-bootstrap-and-cross-validation-and-errors"]], "More interpretations": [[29, "more-interpretations"], [30, "more-interpretations"], [30, "id5"]], "More on Dimensionalities": [[3, "more-on-dimensionalities"]], "More on Rescaling data": [[6, "more-on-rescaling-data"]], "More on Steepest descent": [[30, "more-on-steepest-descent"]], "More on convex functions": [[30, "more-on-convex-functions"]], "More preprocessing": [[29, "more-preprocessing"], [31, "more-preprocessing"]], "Motivation for Adaptive Step Sizes": [[31, "motivation-for-adaptive-step-sizes"]], "Multilayer perceptrons": [[12, "multilayer-perceptrons"]], "Network requirements": [[2, "network-requirements"]], "Neural Networks vs CNNs": [[3, "neural-networks-vs-cnns"]], "Neural networks": [[12, null]], "Non-Convex Problems": [[31, "non-convex-problems"]], "Note about SVD Calculations": [[29, "note-about-svd-calculations"], [30, "note-about-svd-calculations"]], "Note on Scikit-Learn": [[30, "note-on-scikit-learn"]], "Numerical experiments and the covariance, central limit theorem": [[25, "numerical-experiments-and-the-covariance-central-limit-theorem"]], "Numpy and arrays": [[22, "numpy-and-arrays"], [28, "numpy-and-arrays"]], "Numpy examples and Important Matrix and vector handling packages": [[28, "numpy-examples-and-important-matrix-and-vector-handling-packages"]], "Optimization and gradient descent, the central part of any Machine Learning algortithm": [[30, "optimization-and-gradient-descent-the-central-part-of-any-machine-learning-algortithm"]], "Optimization, the central part of any Machine Learning algortithm": [[13, null]], "Optimizing our parameters": [[28, "optimizing-our-parameters"]], "Optimizing our parameters, more details": [[28, "optimizing-our-parameters-more-details"]], "Optimizing the cost function": [[1, "optimizing-the-cost-function"]], "Organizing our data": [[0, "organizing-our-data"], [28, "organizing-our-data"]], "Other Matrix and Vector Operations": [[22, "other-matrix-and-vector-operations"]], "Other Types of Recurrent Neural Networks": [[4, "other-types-of-recurrent-neural-networks"]], "Other courses on Data science and Machine Learning at UiO": [[28, "other-courses-on-data-science-and-machine-learning-at-uio"]], "Other courses on Data science and Machine Learning at UiO, contn": [[28, "other-courses-on-data-science-and-machine-learning-at-uio-contn"]], "Other popular texts": [[28, "other-popular-texts"]], "Other techniques": [[11, "other-techniques"]], "Other types of networks": [[12, "other-types-of-networks"]], "Other ways of visualizing the trees": [[9, "other-ways-of-visualizing-the-trees"]], "Our model for the nuclear binding energies": [[28, "our-model-for-the-nuclear-binding-energies"]], "Overview of first week": [[28, "overview-of-first-week"]], "Overview video on Stochastic Gradient Descent (SGD)": [[31, "overview-video-on-stochastic-gradient-descent-sgd"]], "Own code for Ordinary Least Squares": [[28, "own-code-for-ordinary-least-squares"], [29, "own-code-for-ordinary-least-squares"]], "PCA and scikit-learn": [[11, "pca-and-scikit-learn"]], "Pandas AI": [[28, "pandas-ai"]], "Part a : Ordinary Least Square (OLS) for the Runge function": [[23, "part-a-ordinary-least-square-ols-for-the-runge-function"]], "Part b: Adding Ridge regression for the Runge function": [[23, "part-b-adding-ridge-regression-for-the-runge-function"]], "Part c: Writing your own gradient descent code": [[23, "part-c-writing-your-own-gradient-descent-code"]], "Part d: Including momentum and more advanced ways to update the learning the rate": [[23, "part-d-including-momentum-and-more-advanced-ways-to-update-the-learning-the-rate"]], "Part e: Writing our own code for Lasso regression": [[23, "part-e-writing-our-own-code-for-lasso-regression"]], "Part f: Stochastic gradient descent": [[23, "part-f-stochastic-gradient-descent"]], "Part g: Bias-variance trade-off and resampling techniques": [[23, "part-g-bias-variance-trade-off-and-resampling-techniques"]], "Part h): Cross-validation as resampling techniques, adding more complexity": [[23, "part-h-cross-validation-as-resampling-techniques-adding-more-complexity"]], "Partial Differential Equations": [[2, "partial-differential-equations"]], "Plans for week 35": [[29, "plans-for-week-35"]], "Plans for week 36": [[30, "plans-for-week-36"]], "Plans for week 37, lecture Monday": [[31, "plans-for-week-37-lecture-monday"]], "Plans for week 38, lecture Monday September 15": [[32, "plans-for-week-38-lecture-monday-september-15"]], "Plotting the Histogram": [[32, "plotting-the-histogram"]], "Practical tips": [[13, "practical-tips"], [31, "practical-tips"]], "Practicalities": [[26, "practicalities"], [26, "id1"]], "Preamble: Note on writing reports, using reference material, AI and other tools": [[23, "preamble-note-on-writing-reports-using-reference-material-ai-and-other-tools"]], "Predicting New Points With A Trained Recurrent Neural Network": [[4, "predicting-new-points-with-a-trained-recurrent-neural-network"]], "Preprocessing our data": [[29, "preprocessing-our-data"]], "Prerequisites": [[28, "prerequisites"]], "Prerequisites and background": [[21, "prerequisites-and-background"]], "Prerequisites: Collect and pre-process data": [[3, "prerequisites-collect-and-pre-process-data"]], "Probability Distribution Functions": [[25, "probability-distribution-functions"]], "Program example for gradient descent with Ridge Regression": [[30, "program-example-for-gradient-descent-with-ridge-regression"], [31, "program-example-for-gradient-descent-with-ridge-regression"]], "Program for stochastic gradient": [[13, "program-for-stochastic-gradient"]], "Project 1 on Machine Learning, deadline October 6 (midnight), 2025": [[23, null]], "Properties of PDFs": [[25, "properties-of-pdfs"]], "Pros and cons": [[31, "pros-and-cons"]], "Pros and cons of trees, pros": [[9, "pros-and-cons-of-trees-pros"]], "Python installers": [[21, "python-installers"], [28, "python-installers"]], "RMS prop": [[13, "rms-prop"]], "RMSProp algorithm, taken from Goodfellow et al": [[31, "rmsprop-algorithm-taken-from-goodfellow-et-al"]], "RMSProp: Adaptive Learning Rates": [[31, "rmsprop-adaptive-learning-rates"]], "RMSprop for adaptive learning rate with Stochastic Gradient Descent": [[31, "rmsprop-for-adaptive-learning-rate-with-stochastic-gradient-descent"]], "Random Numbers": [[25, "random-numbers"]], "Random forests": [[10, "random-forests"]], "Randomized PCA": [[11, "randomized-pca"]], "Reading material": [[28, "reading-material"]], "Reading recommendations:": [[29, "reading-recommendations"]], "Reading suggestions week 34": [[28, "reading-suggestions-week-34"]], "Readings and Videos": [[32, "readings-and-videos"]], "Readings and Videos:": [[31, "readings-and-videos"]], "Recurrent neural networks": [[12, "recurrent-neural-networks"]], "Recurrent neural networks: Overarching view": [[4, null]], "Reducing the number of degrees of freedom, overarching view": [[0, "reducing-the-number-of-degrees-of-freedom-overarching-view"], [29, "reducing-the-number-of-degrees-of-freedom-overarching-view"]], "Reformulating the problem": [[2, "reformulating-the-problem"]], "Regression Case": [[10, "regression-case"]], "Regression analysis and resampling methods": [[23, "regression-analysis-and-resampling-methods"]], "Regression analysis, overarching aims": [[28, "regression-analysis-overarching-aims"]], "Regression analysis, overarching aims II": [[28, "regression-analysis-overarching-aims-ii"]], "Regularization": [[1, "regularization"]], "Reminder from last week": [[29, "reminder-from-last-week"]], "Reminder on Newton-Raphson\u2019s method": [[30, "reminder-on-newton-raphson-s-method"]], "Reminder on Statistics": [[6, "reminder-on-statistics"]], "Reminder on different scaling methods": [[31, "reminder-on-different-scaling-methods"]], "Replace or not": [[13, "replace-or-not"], [31, "replace-or-not"]], "Required Technologies": [[21, "required-technologies"]], "Resampling Methods": [[6, null]], "Resampling and the Bias-Variance Trade-off": [[19, "resampling-and-the-bias-variance-trade-off"]], "Resampling approaches can be computationally expensive": [[32, "resampling-approaches-can-be-computationally-expensive"]], "Resampling methods": [[6, "id1"], [32, "resampling-methods"], [32, "id2"]], "Resampling methods: Bootstrap": [[32, "resampling-methods-bootstrap"]], "Resampling methods: Bootstrap approach": [[32, "resampling-methods-bootstrap-approach"]], "Resampling methods: Bootstrap background": [[32, "resampling-methods-bootstrap-background"]], "Resampling methods: Bootstrap steps": [[32, "resampling-methods-bootstrap-steps"]], "Resampling methods: More Bootstrap background": [[32, "resampling-methods-more-bootstrap-background"]], "Residual Error": [[29, "residual-error"], [30, "residual-error"]], "Resources on differential equations and deep learning": [[2, "resources-on-differential-equations-and-deep-learning"]], "Revisiting Ordinary Least Squares": [[30, "revisiting-ordinary-least-squares"]], "Revisiting our Linear Regression Solvers": [[13, "revisiting-our-linear-regression-solvers"]], "Rewriting the Covariance and/or Correlation Matrix": [[29, "rewriting-the-covariance-and-or-correlation-matrix"]], "Rewriting the \\delta-function": [[32, "rewriting-the-delta-function"]], "Rewriting the fitting procedure as a linear algebra problem": [[28, "rewriting-the-fitting-procedure-as-a-linear-algebra-problem"]], "Rewriting the fitting procedure as a linear algebra problem, more details": [[28, "rewriting-the-fitting-procedure-as-a-linear-algebra-problem-more-details"]], "Ridge Regression": [[30, "ridge-regression"]], "Ridge and LASSO Regression": [[29, "ridge-and-lasso-regression"], [30, "ridge-and-lasso-regression"], [30, "id2"]], "Ridge and Lasso Regression": [[5, null], [5, "id1"]], "SGD example": [[31, "sgd-example"]], "SGD vs Full-Batch GD: Convergence Speed and Memory Comparison": [[31, "sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison"]], "SVD analysis": [[30, "svd-analysis"]], "Same code but now with momentum gradient descent": [[13, "same-code-but-now-with-momentum-gradient-descent"], [31, "same-code-but-now-with-momentum-gradient-descent"], [31, "id3"], [31, "id4"]], "Schedule first week": [[28, "schedule-first-week"]], "Schematic Regression Procedure": [[9, "schematic-regression-procedure"]], "Second moment of the gradient": [[31, "second-moment-of-the-gradient"]], "September 15-19": [[19, "september-15-19"]], "Setting up the Back propagation algorithm": [[12, "setting-up-the-back-propagation-algorithm"]], "Setting up the Matrix to be inverted": [[29, "setting-up-the-matrix-to-be-inverted"], [30, "setting-up-the-matrix-to-be-inverted"]], "Setting up the network using Autograd; The full program": [[2, "setting-up-the-network-using-autograd-the-full-program"]], "Similar (second order function now) problem but now with AdaGrad": [[13, "similar-second-order-function-now-problem-but-now-with-adagrad"], [31, "similar-second-order-function-now-problem-but-now-with-adagrad"]], "Simple Python Code to read in Data and perform Classification": [[9, "simple-python-code-to-read-in-data-and-perform-classification"]], "Simple case": [[29, "simple-case"], [30, "simple-case"]], "Simple code for solving the above problem": [[30, "simple-code-for-solving-the-above-problem"]], "Simple example code": [[31, "simple-example-code"]], "Simple example to illustrate Ordinary Least Squares, Ridge and Lasso Regression": [[30, "simple-example-to-illustrate-ordinary-least-squares-ridge-and-lasso-regression"]], "Simple geometric interpretation": [[30, "simple-geometric-interpretation"]], "Simple linear regression model using scikit-learn": [[0, "simple-linear-regression-model-using-scikit-learn"], [28, "simple-linear-regression-model-using-scikit-learn"]], "Simple one-dimensional second-order polynomial": [[18, "simple-one-dimensional-second-order-polynomial"]], "Simple program": [[30, "simple-program"], [31, "simple-program"]], "Slightly different approach": [[31, "slightly-different-approach"]], "Sneaking in automatic differentiation using Autograd": [[31, "sneaking-in-automatic-differentiation-using-autograd"]], "Software and needed installations": [[23, "software-and-needed-installations"], [28, "software-and-needed-installations"]], "Solving Differential Equations with Deep Learning": [[2, null]], "Solving the one dimensional Poisson equation": [[2, "solving-the-one-dimensional-poisson-equation"]], "Solving the wave equation with Neural Networks": [[2, "solving-the-wave-equation-with-neural-networks"]], "Some famous Matrices": [[22, "some-famous-matrices"]], "Some simple problems": [[13, "some-simple-problems"], [30, "some-simple-problems"]], "Some useful matrix and vector expressions": [[29, "some-useful-matrix-and-vector-expressions"]], "Splitting our Data in Training and Test data": [[0, "splitting-our-data-in-training-and-test-data"], [29, "splitting-our-data-in-training-and-test-data"]], "Standard Approach based on the Normal Distribution": [[32, "standard-approach-based-on-the-normal-distribution"]], "Standard steepest descent": [[13, "standard-steepest-descent"]], "Statistical analysis": [[32, "statistical-analysis"]], "Statistical analysis and optimization of data": [[21, "statistical-analysis-and-optimization-of-data"], [28, "statistical-analysis-and-optimization-of-data"]], "Steepest descent": [[13, "steepest-descent"], [30, "steepest-descent"]], "Stochastic Gradient Descent": [[31, "stochastic-gradient-descent"]], "Stochastic Gradient Descent (SGD)": [[13, "stochastic-gradient-descent-sgd"], [31, "stochastic-gradient-descent-sgd"]], "Stochastic variables and the main concepts, the discrete case": [[25, "stochastic-variables-and-the-main-concepts-the-discrete-case"]], "Strongly Convex Case": [[31, "strongly-convex-case"]], "Summing up": [[32, "summing-up"]], "Support Vector Machines, overarching aims": [[8, null]], "Systematic reduction": [[3, "systematic-reduction"]], "Teachers": [[28, "teachers"]], "Teachers and Grading": [[26, null]], "Teaching Assistants Fall semester 2023": [[26, "teaching-assistants-fall-semester-2023"]], "Tentative deadllines for projects": [[26, "tentative-deadllines-for-projects"]], "Testing the Means Squared Error as function of Complexity": [[0, "testing-the-means-squared-error-as-function-of-complexity"], [29, "testing-the-means-squared-error-as-function-of-complexity"]], "Textbooks": [[27, null]], "The Algorithm before theorem": [[11, "the-algorithm-before-theorem"]], "The Breast Cancer Data, now with Keras": [[1, "the-breast-cancer-data-now-with-keras"]], "The CART algorithm for Classification": [[9, "the-cart-algorithm-for-classification"]], "The CART algorithm for Regression": [[9, "the-cart-algorithm-for-regression"]], "The CIFAR01 data set": [[3, "the-cifar01-data-set"]], "The Central Limit Theorem": [[32, "the-central-limit-theorem"]], "The Hessian matrix": [[30, "the-hessian-matrix"], [31, "the-hessian-matrix"]], "The Hessian matrix for Ridge Regression": [[30, "the-hessian-matrix-for-ridge-regression"], [31, "the-hessian-matrix-for-ridge-regression"]], "The Jacobian": [[29, "the-jacobian"]], "The MNIST dataset again": [[3, "the-mnist-dataset-again"]], "The OLS case": [[30, "the-ols-case"]], "The RELU function family": [[1, "the-relu-function-family"]], "The Ridge case": [[30, "the-ridge-case"]], "The SVD, a Fantastic Algorithm": [[29, "the-svd-a-fantastic-algorithm"], [30, "the-svd-a-fantastic-algorithm"]], "The Softmax function": [[1, "the-softmax-function"]], "The \\chi^2 function": [[0, "the-chi-2-function"], [28, "the-chi-2-function"], [28, "id4"], [28, "id5"], [28, "id6"], [28, "id7"], [28, "id8"]], "The bias-variance tradeoff": [[6, "the-bias-variance-tradeoff"], [32, "the-bias-variance-tradeoff"]], "The code for solving the ODE": [[2, "the-code-for-solving-the-ode"]], "The complete code with a simple data set": [[29, "the-complete-code-with-a-simple-data-set"]], "The cost/loss function": [[29, "the-cost-loss-function"]], "The course has two central parts": [[21, "the-course-has-two-central-parts"]], "The derivative of the cost/loss function": [[30, "the-derivative-of-the-cost-loss-function"], [31, "the-derivative-of-the-cost-loss-function"]], "The equations": [[30, "the-equations"]], "The equations for ordinary least squares": [[29, "the-equations-for-ordinary-least-squares"]], "The first Case": [[30, "the-first-case"]], "The gradient step": [[31, "the-gradient-step"]], "The ideal": [[30, "the-ideal"]], "The logistic function": [[7, "the-logistic-function"]], "The mean squared error and its derivative": [[29, "the-mean-squared-error-and-its-derivative"]], "The moons example": [[8, "the-moons-example"]], "The multilayer perceptron (MLP)": [[12, "the-multilayer-perceptron-mlp"]], "The network with one input layer, specified number of hidden layers, and one output layer": [[2, "the-network-with-one-input-layer-specified-number-of-hidden-layers-and-one-output-layer"]], "The plethora of machine learning algorithms/methods": [[28, "the-plethora-of-machine-learning-algorithms-methods"]], "The same example but now with cross-validation": [[32, "the-same-example-but-now-with-cross-validation"]], "The sensitiveness of the gradient descent": [[30, "the-sensitiveness-of-the-gradient-descent"]], "The singular value decomposition": [[5, "the-singular-value-decomposition"], [29, "the-singular-value-decomposition"], [30, "the-singular-value-decomposition"]], "The two-dimensional case": [[8, "the-two-dimensional-case"]], "Theoretical Convergence Speed and convex optimization": [[31, "theoretical-convergence-speed-and-convex-optimization"]], "Time decay rate": [[31, "time-decay-rate"]], "To our real data: nuclear binding energies. Brief reminder on masses and binding energies": [[28, "to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies"]], "Topics covered in this course: Statistical analysis and optimization of data": [[28, "topics-covered-in-this-course-statistical-analysis-and-optimization-of-data"]], "Towards the PCA theorem": [[11, "towards-the-pca-theorem"]], "Train and test datasets": [[1, "train-and-test-datasets"]], "Two-dimensional Objects": [[3, "two-dimensional-objects"]], "Type of problem": [[2, "type-of-problem"]], "Types of Machine Learning": [[28, "types-of-machine-learning"]], "Understanding what happens": [[32, "understanding-what-happens"]], "Use the books!": [[19, "use-the-books"]], "Useful Python libraries": [[21, "useful-python-libraries"], [28, "useful-python-libraries"]], "Using Autograd": [[13, "using-autograd"]], "Using forward Euler to solve the ODE": [[2, "using-forward-euler-to-solve-the-ode"]], "Using gradient descent methods, limitations": [[13, "using-gradient-descent-methods-limitations"], [30, "using-gradient-descent-methods-limitations"], [31, "using-gradient-descent-methods-limitations"]], "Various steps in cross-validation": [[32, "various-steps-in-cross-validation"]], "Visualization": [[1, "visualization"], [1, "id1"]], "Visualizing the Tree, Classification": [[9, "visualizing-the-tree-classification"]], "Week 34: Introduction to the course, Logistics and Practicalities": [[28, null]], "Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression": [[29, null]], "Week 36: Linear Regression and Gradient descent": [[30, null]], "Week 37: Gradient descent methods": [[31, null]], "Week 38: Statistical analysis, bias-variance tradeoff and resampling methods": [[32, null]], "What Is Generative Modeling?": [[28, "what-is-generative-modeling"]], "What does it mean?": [[29, "what-does-it-mean"], [30, "what-does-it-mean"]], "What is Machine Learning?": [[0, "what-is-machine-learning"]], "What is a good model?": [[0, "what-is-a-good-model"], [28, "what-is-a-good-model"]], "What is a good model? Can we define it?": [[28, "what-is-a-good-model-can-we-define-it"]], "When do we stop?": [[31, "when-do-we-stop"]], "Which activation function should I use?": [[1, "which-activation-function-should-i-use"]], "Why Combine Momentum and RMSProp?": [[31, "why-combine-momentum-and-rmsprop"]], "Why Linear Regression (aka Ordinary Least Squares and family)": [[28, "why-linear-regression-aka-ordinary-least-squares-and-family"]], "Why resampling methods": [[32, "why-resampling-methods"]], "Why resampling methods ?": [[32, "id1"]], "Wisconsin Cancer Data": [[7, "wisconsin-cancer-data"]], "With Lasso Regression": [[30, "with-lasso-regression"]], "Wrapping it up": [[32, "wrapping-it-up"]], "Writing Our First Generative Adversarial Network": [[4, "writing-our-first-generative-adversarial-network"]], "Writing our own PCA code": [[11, "writing-our-own-pca-code"]], "Writing the Cost Function": [[30, "writing-the-cost-function"]], "XGBoost: Extreme Gradient Boosting": [[10, "xgboost-extreme-gradient-boosting"]], "Yet another Example": [[30, "yet-another-example"]], "a) Expression for Ridge regression": [[17, "a-expression-for-ridge-regression"]], "scikit-learn implementation": [[1, "scikit-learn-implementation"]]}, "docnames": ["chapter1", "chapter10", "chapter11", "chapter12", "chapter13", "chapter2", "chapter3", "chapter4", "chapter5", "chapter6", "chapter7", "chapter8", "chapter9", "chapteroptimization", "clustering", "exercisesweek34", "exercisesweek35", "exercisesweek36", "exercisesweek37", "exercisesweek38", "exercisesweek39", "intro", "linalg", "project1", "schedule", "statistics", "teachers", "textbooks", "week34", "week35", "week36", "week37", "week38"], "envversion": {"sphinx": 62, "sphinx.domains.c": 3, "sphinx.domains.changeset": 1, "sphinx.domains.citation": 1, "sphinx.domains.cpp": 9, "sphinx.domains.index": 1, "sphinx.domains.javascript": 3, "sphinx.domains.math": 2, "sphinx.domains.python": 4, "sphinx.domains.rst": 2, "sphinx.domains.std": 2, "sphinx.ext.intersphinx": 1}, "filenames": ["chapter1.ipynb", "chapter10.ipynb", "chapter11.ipynb", "chapter12.ipynb", "chapter13.ipynb", "chapter2.ipynb", "chapter3.ipynb", "chapter4.ipynb", "chapter5.ipynb", "chapter6.ipynb", "chapter7.ipynb", "chapter8.ipynb", "chapter9.ipynb", "chapteroptimization.ipynb", "clustering.ipynb", "exercisesweek34.ipynb", "exercisesweek35.ipynb", "exercisesweek36.ipynb", "exercisesweek37.ipynb", "exercisesweek38.ipynb", "exercisesweek39.ipynb", "intro.md", "linalg.ipynb", "project1.ipynb", "schedule.md", "statistics.ipynb", "teachers.md", "textbooks.md", "week34.ipynb", "week35.ipynb", "week36.ipynb", "week37.ipynb", "week38.ipynb"], "indexentries": {}, "objects": {}, "objnames": {}, "objtypes": {}, "terms": {"": [0, 1, 2, 3, 4, 5, 6, 7, 9, 11, 12, 13, 15, 16, 17, 19, 21, 22, 23, 25, 26, 28, 29], "0": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 26, 28, 29, 30, 31, 32], "00": [0, 1, 5, 11, 28, 29], "000": [1, 3], "000000": [], "00000000e": [], "001": [2, 8, 13, 30, 31], "004": 5, "004113634617443131": 29, "004113634617443139": 29, "00411363461744314": 29, "004113634617443147": 29, "005b82": [], "00622f": [], "00727646693": [0, 28], "0072b2": [], "00749c": [], "008561": [], "0086649156": [0, 28], "00e0e0": [], "01": [0, 1, 2, 5, 9, 11, 13, 17, 27, 28, 29, 31], "010726": [], "0110": 25, "01719003e": [], "02": [0, 4, 7, 12, 28], "02334824": [], "023b95": [], "024c1a": [], "02857": 4, "02f": 6, "03077640549": 4, "03097597e": [], "031": 5, "04": 11, "0458": 9, "05": [4, 6], "0550ae": [], "062292565": 4, "062435": [], "06730814": [], "07": [], "0713": [0, 28], "07285": 3, "08": 25, "08078025e": [], "080808": [], "08336233266": 4, "08376632": 29, "083766322923899": 29, "0837663229239043": 29, "0917": 9, "0969da4a": [], "0d1117": [], "0n": [0, 28], "0x113e21950": 17, "1": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 24, 25, 26, 27, 28, 30, 31, 32], "10": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 18, 19, 22, 24, 25, 26, 28, 29, 30, 31, 32], "100": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 25, 26, 28, 29, 30, 31, 32], "1000": [0, 1, 2, 4, 5, 8, 11, 13, 14, 18, 19, 21, 25, 28, 30, 31], "10000": [2, 5, 6, 10, 11, 13, 25, 32], "100000": 8, "10001": 10, "1001": 25, "1002": 25, "1003": 25, "1005": 25, "1007": 32, "1009": 25, "101": 16, "1011": 25, "1013": 25, "1013904243": 25, "1015": 25, "102": 16, "1023": 25, "1024": 3, "1026": 25, "1027": 25, "103": 1, "1030": 25, "1037": 25, "1038": 25, "1040": 25, "1047": 25, "107": 16, "108": [], "10th": 9, "10x": [0, 28], "11": [0, 2, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 22, 23, 25, 27, 28, 29, 30, 31, 32], "110": [], "1100": 25, "1101": 25, "111": [1, 7, 12], "112": 16, "11340253": [], "11590451": [], "116": 16, "116329": [], "116633": [], "117": 16, "118": 16, "12": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 18, 22, 25, 27, 28, 29, 30, 31, 32], "120": 3, "121": [8, 9, 10, 16], "1215pm": [26, 28], "122": [8, 9, 10], "124": [0, 28], "125": 16, "127": [4, 16], "128": [3, 4, 13, 31], "129": 16, "1298": 9, "12pm": [26, 28], "13": [0, 2, 9, 12, 22, 25, 28], "131": 16, "133": 7, "135": 16, "136": 16, "14": [0, 2, 4, 6, 8, 9, 10, 12, 22, 25, 27, 29, 32], "141": 16, "1412": 31, "141414": [], "143": 16, "1446729567": 4, "149": 16, "14g": [6, 32], "15": [0, 2, 4, 6, 7, 8, 9, 12, 13, 23, 25, 28, 30, 31], "150": [4, 8], "152": 16, "153760": [], "156": 16, "157": [], "158": [], "159": 16, "15g": [6, 32], "15pm": 28, "16": [1, 2, 3, 4, 5, 8, 9, 10, 25, 28, 30, 32], "160": 16, "1603": 3, "161": 16, "162": 16, "16231451": 4, "163": 16, "16384": 3, "164": 16, "167": 16, "17": [1, 2, 8, 25], "172": 16, "173": 16, "175": 32, "176": 16, "178": 16, "179": 16, "1797": 1, "18": [2, 6, 7, 8, 9, 10, 25, 28, 32], "1807": 4, "181036": [], "18392847": [], "18c1c4": [], "19": [2, 25, 28, 32], "192": 32, "1940": [], "1943": 12, "19569961": 29, "19680801": [], "1970": [22, 28], "1973": 9, "1979": [6, 32], "1_1": 12, "1_2": 12, "1_3": 12, "1cm": [0, 8, 10, 25, 28], "1d": [1, 2, 3], "1e": [2, 4, 13, 14, 31], "1e10": 14, "1e1e1": [], "1e4": 6, "1f": 1, "1k": 22, "1n": [0, 28], "1x": [0, 28], "2": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 25, 27, 31, 32], "20": [0, 1, 2, 6, 7, 8, 16, 17, 25, 26, 28, 29, 30, 31, 32], "200": [0, 2, 3, 4, 8, 9, 10], "2000": [0, 29], "2001": [], "2004": [13, 30], "2006": 27, "2007": [], "20072279": [], "2008": [28, 31], "2009": [], "2010": 1, "2011": [1, 31], "2012": 31, "2013": [], "2014": [4, 31], "2015": 1, "2016": [0, 28], "2018": [0, 6, 29, 32], "2019": [], "2020": [], "2021": [6, 14, 29, 31], "2022": 28, "2024": 32, "2025": [18, 28, 29, 30, 31, 32], "21": [0, 1, 5, 7, 9, 12, 22, 28, 29, 30], "2116753732": 4, "215pm": [26, 28], "2167072": [], "22": [0, 1, 5, 12, 13, 22, 28, 29, 30], "221": 8, "225": 4, "22948497": [], "23": [1, 12, 22], "24": [0, 1, 22, 28], "242424": [], "24292f": [], "25": [2, 3, 4, 5, 6, 8, 9, 11, 29], "250": [2, 4, 7, 9], "25000": [], "250154": [], "252124": [], "253775": [], "255": 3, "256": [4, 31], "25x": 23, "26": [], "26303845": [], "264": [], "265": [], "265109911": 4, "266": [], "269": [], "27": 1, "270": [], "278": [30, 31], "27n_": 25, "28": [1, 3, 4], "283": [30, 31], "2830637392": 4, "2861": 25, "2873": 9, "2882": 25, "2886": 25, "2890": [0, 28], "2892": 25, "29": 29, "2915": 25, "2931": 28, "29364655": [], "294399745619595": [], "296247": [], "2968": 28, "2980": 28, "298273": [], "298375": [], "2990": 28, "2_": 12, "2_1": 12, "2_2": 12, "2_3": 12, "2_i": 12, "2_m": [6, 25, 32], "2_t": 13, "2_x": 25, "2a": 17, "2a1968": [], "2b": 25, "2b2b2b": [], "2c8f433990d1": 31, "2cm": 8, "2d": [1, 3, 11, 12, 21, 28], "2e": [6, 32], "2f": [0, 7, 9, 10, 11, 12, 28], "2g": 2, "2g_i": 2, "2k": 3, "2m": [6, 32], "2mvizaqfst8": 29, "2n": [0, 2, 3, 28, 29], "2nd": 9, "2p": 25, "2pt": 4, "2x": [0, 3, 8, 13, 28], "2x_ix_jy_iy_j": 8, "2x_j": 8, "2y_i": 10, "2y_j": 8, "3": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 24, 25, 26, 28, 30, 31, 32], "30": [0, 1, 4, 6, 7, 10, 13, 26, 31, 32], "30000": [0, 28], "3072": 3, "31": [12, 22, 25], "315": [6, 29, 31], "3155": [0, 5, 6, 29, 30, 31, 32], "32": [3, 4, 6, 12, 13, 22, 25, 31], "3200": 1, "3250": 1, "3297": [], "33": [12, 22, 26], "3303": [], "3310": [], "332331": [], "333": 7, "3331": [], "3337": [], "34": 22, "3436": [0, 28], "3437": [0, 28], "35": [0, 6, 23, 28, 30, 31], "3581341341": 4, "359": [5, 30], "36": [0, 5, 6, 18, 23, 25], "37": [23, 30, 32], "370782966": 4, "38": [23, 25], "387": 32, "39": [0, 26, 28], "3d": [2, 3, 4, 6, 13, 16, 32], "3d73a9": [], "3f": [1, 3, 9], "3n": 22, "3x": [2, 8], "3x_i": 2, "3y": 8, "4": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 23, 25, 28, 30, 31, 32], "40": [1, 6, 26, 28, 32], "400": 4, "4000": 28, "40008b9a5380fcacce3976bf7c08af5b": 31, "4050": [27, 28], "41": 22, "4155": [2, 15], "41589548": [], "42": [1, 4, 8, 9, 10, 22], "43": [0, 7, 22], "4310": 28, "436462435": 4, "437a6b": [], "44": [0, 22, 30, 31], "45": [26, 28], "46": [26, 28], "462": 7, "47": [26, 28], "473d18": [], "479465113": 4, "47958494": [], "48": [], "48257387": [26, 28], "49": [5, 6, 11], "49152": 3, "4940954": [0, 28], "4990": 25, "4992": 25, "4997": 25, "4c4b4be8": [], "4c4c7f": [9, 10], "4d": 3, "4f": 6, "4pm": [26, 28], "4y": 8, "4y_i": 10, "5": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 22, 23, 25, 28, 29, 30, 31, 32], "50": [1, 2, 3, 4, 6, 7, 8, 10, 13, 28, 29, 31, 32], "500": [1, 3, 4, 6, 9, 10, 13, 31, 32], "5018": 25, "506": [], "507d50": [9, 10], "50j": 13, "50x10": 1, "51": 10, "510": 1, "512132": [], "515151": [], "5177783846": 4, "53": 9, "5391cf": [], "54": [6, 25], "5411205": [], "54894451": [], "55": 1, "56": 1, "56536": [0, 28], "569": 1, "57": [0, 8, 26, 28], "571": [5, 30], "576": 32, "58": [10, 26, 28], "58a6ff70": [], "591317992": 4, "5ca7e4": [], "5cm": 25, "5f": [8, 31], "5x": [8, 18], "5y": 8, "6": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 18, 22, 25, 26, 28, 29, 30, 31, 32], "60": [1, 3], "60000": 4, "6019067271": 4, "606439": [], "622cbc": [], "625": 7, "63": 1, "64": [1, 3, 4, 13, 22, 28, 31], "64x50": 1, "65": [1, 8, 9], "66666691": [], "66707b": [], "66ccee": [], "66e9ec": [], "6730c5": [], "6887363571": 4, "69": [16, 25], "69069n_": 25, "691": [], "6980": 31, "6e7681": [], "6e7781": [], "6f98b3": [], "6n_": 25, "7": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 22, 23, 25, 27, 28, 29, 31, 32], "70": [1, 7], "702c00": [], "70653767": 4, "71": 1, "724": 3, "72f088": [], "73": [], "7304881": [], "737373": [], "75": [5, 6, 8, 11, 32], "76": [26, 28], "765": 7, "77": [26, 28], "7718": 9, "7782028952": 4, "77893972": [], "78": [], "797979": [], "7998f2": [], "79c0ff": [], "7d7d58": [9, 10], "7ee787": [], "7f4707": [], "8": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 18, 19, 22, 25, 26, 28, 30], "80": [0, 1, 5, 8, 17, 29], "800": [4, 7], "8045e5": [], "81": 1, "815am": [26, 28], "81b19b": [], "8250df": [], "84858": 32, "85": 1, "8702784034": 4, "8786ac": [], "88": 28, "8a4600": [], "8b949e": [], "8c8c8c": [], "8f": [6, 32], "8g": [6, 32], "8n": 22, "8x8": 1, "9": [0, 1, 2, 4, 5, 6, 7, 8, 9, 11, 12, 13, 22, 25, 28, 31], "90": 1, "9040": 9, "91": [26, 28], "912583": [], "91cbff": [], "92": [26, 28], "93": 16, "931": [0, 28], "933": [5, 30], "937": 25, "938": 25, "939": [0, 25, 28], "94": 25, "95": [1, 11, 32], "953800": [], "954": 25, "955820c21e8b": 4, "96": [6, 32], "960": 25, "961": 25, "962": 25, "9649652536": 4, "96611194e": [], "974eb7": [], "978": 32, "9780387310732": 27, "9780387848570": 27, "9781098134174": 28, "9781492032632": 27, "9781801819312": 28, "97898392": 29, "98": [0, 1, 16], "985": 25, "986": 25, "98661b": [], "989": 25, "9898ff": [9, 10], "99": [13, 16, 31, 32], "991": 25, "992": 25, "993": 25, "996": 5, "996b00": [], "999": [9, 25, 31], "999999": [], "9e86c8": [], "9e8741": [], "9f4e55": [], "9x": 6, "9y": 6, "A": [2, 3, 5, 6, 7, 10, 11, 12, 13, 15, 16, 19, 20, 21, 22, 24, 25, 26, 27, 29, 30, 31], "AND": 2, "AS": [], "AT": [], "And": [0, 3, 4, 5, 6, 9, 13, 20, 21, 23, 25, 30], "As": [0, 1, 2, 3, 4, 5, 6, 8, 10, 12, 13, 15, 16, 22, 23, 25, 28, 29, 30, 31, 32], "At": [0, 4, 6, 13, 20, 28, 31], "BE": [0, 28], "BUT": [], "BY": [], "Be": [2, 18, 21, 28], "Being": 13, "But": [0, 1, 2, 3, 5, 6, 9, 10, 16, 25, 29, 32], "By": [0, 3, 5, 6, 12, 13, 17, 19, 22, 28, 29, 30, 31, 32], "FOR": [], "For": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "IF": [6, 29, 31], "IN": 27, "If": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 18, 21, 22, 23, 25, 28, 29, 30, 31, 32], "In": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "Ising": [5, 12, 29, 30], "It": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "Its": [1, 2, 4, 11], "NO": [], "NOT": [], "No": [6, 9, 28, 29, 31], "Not": [0, 1, 5, 6, 29, 30, 31, 32], "OF": [], "ON": [], "OR": 25, "Of": 25, "On": [0, 3, 23, 25, 26, 27, 28, 31, 32], "One": [0, 1, 3, 4, 5, 6, 7, 8, 11, 12, 13, 17, 25, 29, 30, 31, 32], "Or": [0, 1, 6, 28], "SUCH": [], "Such": [0, 6, 12, 16, 25, 31, 32], "THE": [], "TO": [], "That": [0, 5, 7, 10, 11, 12, 14, 23, 25, 28, 32], "The": [4, 10, 13, 14, 16, 17, 18, 19, 20, 22, 23, 24, 25, 26, 27], "Then": [0, 1, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 22, 28, 30, 31, 32], "There": [0, 3, 4, 5, 6, 8, 9, 11, 12, 14, 15, 22, 23, 25, 26, 28, 29, 30, 31], "These": [0, 3, 4, 5, 8, 9, 10, 11, 12, 13, 14, 17, 18, 22, 23, 25, 26, 28, 29, 30, 31], "To": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 20, 22, 25, 29, 30, 31, 32], "WITH": [], "With": [0, 5, 6, 8, 9, 10, 11, 12, 14, 16, 19, 22, 23, 25, 28, 29, 32], "_": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 18, 19, 22, 23, 28, 29, 30, 31, 32], "_0": [5, 8, 10, 11, 13, 29, 30], "_1": [2, 5, 6, 8, 10, 11, 12, 13, 14, 22, 29, 30, 31], "_2": [2, 5, 8, 11, 12, 13, 22, 29, 31], "_3": 22, "_4": 22, "_9": [13, 31], "__array_finalize__": [], "__class__": 10, "__doc__": [6, 32], "__future__": [8, 9], "__getattribute__": [], "__import__": [], "__init__": 1, "__main__": 2, "__name__": [2, 10], "__new__": [], "__path__": [], "_auto1": [2, 3, 4, 5, 6, 7, 12, 13, 22, 25, 29, 30], "_auto10": [6, 12], "_auto11": 6, "_auto12": 6, "_auto2": [2, 3, 4, 5, 6, 12, 13, 22, 25], "_auto3": [3, 4, 5, 6, 12, 13, 22], "_auto4": [4, 6, 12, 13, 22], "_auto5": [4, 6, 12, 13, 22], "_auto6": [4, 6, 12, 22], "_auto7": [4, 6, 12, 22], "_auto8": [6, 12], "_auto9": [6, 12], "_build": [0, 21, 23, 27, 28], "_c": 1, "_center": [], "_compile_transl": [], "_compon": 11, "_data": [], "_depth": 9, "_export": [15, 16, 19], "_fraction": 9, "_i": [0, 1, 2, 5, 6, 7, 8, 11, 12, 13, 19, 23, 28, 29, 30, 31, 32], "_j": [0, 1, 2, 3, 5, 6, 8, 13, 19, 23, 29, 30, 31, 32], "_k": [13, 30, 31], "_l": 12, "_lambda": 6, "_leaf": 9, "_m": 10, "_mask": [], "_multilayer_perceptron": [], "_n": [2, 5, 8, 11, 13, 29, 30, 31], "_node": 9, "_norm": [], "_p": [5, 8, 29, 30], "_parse_numpydoc_see_also_sect": [], "_pydevd_bundl": [], "_ratio": 11, "_sampl": 9, "_split": [6, 9, 23], "_t": [13, 31], "_test": [6, 23], "_varianc": 11, "_weight": 9, "a0": 3, "a0111f": [], "a0faa0": [9, 10], "a1": [0, 28], "a11": [], "a12236": [], "a2": [0, 28], "a25e53": [], "a2bffc": [], "a3": [0, 28], "a4": [0, 28], "a5d6ff": [], "a_": [0, 1, 16, 22, 28, 29], "a_0": [0, 28], "a_1a": [0, 28], "a_2a": [0, 28], "a_3": [0, 28], "a_3a": [0, 28], "a_4": [0, 28], "a_4a": [0, 28], "a_h": 1, "a_i": [0, 1, 2, 12, 28], "a_j": [1, 12], "a_k": [0, 1, 12], "aa": [], "aaa": [], "aaron": 27, "ab": [0, 2, 5, 13, 14, 28, 29, 31], "ab6369": [], "ab_channel": 21, "abandon": 1, "abe338": [], "abid": 25, "abil": [0, 10], "abl": [0, 1, 4, 5, 6, 7, 10, 12, 13, 16, 18, 20, 23, 29, 30, 31], "about": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 19, 20, 21, 22, 23, 26, 31, 32], "abov": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 25, 27, 28, 29, 31, 32], "abovement": [6, 23, 28, 32], "abscissa": [13, 30], "absent": 31, "absolut": [0, 2, 5, 6, 13, 28, 29, 30, 32], "absorb": [29, 30], "abstract": [1, 31], "abund": 31, "ac": [], "acceler": [13, 31], "accept": [0, 3, 6, 9, 23, 29, 31], "access": [3, 11, 25, 28, 31], "accid": [4, 6, 32], "accompani": [0, 28, 29], "accomplish": [8, 9, 13, 31], "accord": [0, 1, 2, 5, 6, 9, 12, 13, 14, 25, 28, 30, 31, 32], "accordingli": 11, "account": [0, 3, 5, 13, 15, 16, 20, 25, 28, 31], "accumul": [12, 13, 25, 31], "accur": [0, 3, 4, 6, 10, 13, 31, 32], "accuraci": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 12, 28, 29, 30], "accuracy_scor": [0, 1, 10, 28], "accuracy_score_numpi": 1, "achiev": [0, 1, 5, 6, 8, 12, 22, 28, 31, 32], "aco": 25, "acquaint": 21, "acquir": [1, 21, 28], "acr": [], "across": [1, 3, 6, 9, 17, 21, 28, 32], "act": [1, 3, 22, 31], "action": 25, "activ": [0, 2, 3, 4, 9, 15, 24, 26, 28, 31], "activest": [], "actual": [0, 1, 4, 5, 6, 8, 11, 15, 16, 18, 22, 25, 28, 29, 30, 31, 32], "ad": [1, 3, 4, 5, 8, 13, 15, 16, 22, 30, 31, 32], "ada_clf": 10, "adaboostclassifi": 10, "adadelta": [13, 31], "adagrad": [23, 32], "adam": [1, 3, 4, 23, 28, 32], "adapt": [4, 6, 13, 17, 27, 30, 32], "add": [0, 1, 2, 3, 4, 5, 6, 8, 10, 11, 12, 15, 16, 17, 18, 20, 25, 26, 28, 29, 30, 31, 32], "add6ff": [], "add_": [], "add_subplot": [1, 7, 12, 14], "addendum": 5, "addeventlisten": [], "addit": [0, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 15, 21, 22, 23, 25, 26, 27, 28, 29, 32], "addition": [12, 13, 30, 31], "address": [1, 9, 11, 13, 28, 31], "adjac": [3, 12], "adjoint": [5, 29], "adjust": [0, 5, 12, 13, 30, 31], "admir": [0, 28], "advanc": [4, 6, 12, 27, 28, 31, 32], "advantag": [1, 3, 5, 6, 10, 13, 19, 22, 30, 31, 32], "adversari": 28, "advis": [], "afecionado": 28, "affect": [3, 15, 19], "affin": [0, 3, 8, 11, 29], "afford": 3, "aficionado": 28, "aforement": 14, "african": [], "after": [0, 1, 2, 4, 5, 6, 9, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "afterward": [0, 28], "ag": [0, 7, 28, 29], "ag_0": 2, "again": [0, 1, 4, 5, 6, 7, 8, 10, 11, 12, 13, 23, 25, 28, 29, 30, 32], "against": [1, 4, 7, 10], "agegroup": 7, "agegroupmean": 7, "aggreg": [9, 10, 31], "agorithm": 10, "agre": [5, 6, 25, 29, 30, 31, 32], "agreement": [13, 31], "ahead": 9, "ai": [0, 27], "aid": [11, 20, 31], "aim": [0, 1, 4, 6, 7, 11, 14, 16, 17, 19, 20, 21, 22, 23, 29, 32], "ainv": 5, "airplan": 3, "aka": 5, "al": [0, 2, 4, 16, 17, 20, 27, 28, 29, 30, 32], "alarm": [5, 7], "aldo": 29, "algebra": [0, 3, 5, 13, 21, 29, 30, 32], "algorithm": [0, 1, 2, 4, 5, 6, 7, 8, 13, 14, 16, 21, 22, 23, 25, 27, 32], "align": [0, 2, 5, 6, 7, 8, 13, 25, 28, 29, 30, 32], "all": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15, 18, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32], "allevi": [1, 13, 30], "alloc": [3, 22], "allow": [0, 1, 2, 3, 5, 6, 8, 10, 13, 15, 21, 22, 23, 28, 29, 30, 31, 32], "almost": [0, 1, 6, 8, 11, 13, 25, 30, 31, 32], "alon": [2, 9, 31], "along": [2, 3, 4, 5, 6, 9, 10, 11, 15, 20, 21, 22, 28, 29, 30, 32], "alpha": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 13, 14, 25, 28, 29, 30, 31, 32], "alpha_": [10, 31], "alpha_0": 3, "alpha_1": 3, "alpha_2": 3, "alpha_i": [3, 13], "alpha_k": 13, "alpha_m": 10, "alpha_n": 3, "alpha_opt": 13, "alreadi": [2, 3, 4, 5, 6, 10, 12, 15, 21, 22, 25, 28, 29, 30], "also": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "alter": 1, "altern": [0, 1, 4, 5, 6, 8, 9, 11, 13, 15, 18, 22, 23, 28, 29, 31, 32], "although": [0, 1, 5, 6, 8, 10, 13, 16, 19, 20, 28, 31, 32], "alwai": [0, 3, 5, 6, 12, 13, 16, 19, 23, 25, 28, 29, 30, 31, 32], "am": 4, "ame2016": [0, 28], "american": [], "amjith": [], "among": [0, 3, 5, 9, 10, 12, 22, 28, 29], "amongst": [5, 32], "amount": [0, 1, 3, 4, 6, 8, 10, 14, 21, 32], "an": [1, 2, 3, 5, 6, 7, 8, 9, 11, 12, 13, 14, 16, 17, 18, 19, 21, 22, 23, 25, 26, 27, 29, 30, 31, 32], "an_": 25, "anaconda": [0, 1, 21, 23, 28], "analogi": 13, "analys": [6, 32], "analysi": [1, 3, 4, 7, 14, 19, 22, 27, 31], "analyt": [2, 3, 5, 6, 7, 12, 13, 17, 21, 23, 28, 29, 30, 31, 32], "analyz": [0, 1, 3, 4, 5, 6, 16, 23, 25, 29, 30, 31], "andrew": 1, "angl": [0, 3, 9, 29, 31], "anharmon": 3, "ani": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 14, 15, 16, 19, 25, 28, 29, 31, 32], "anim": [4, 12], "ann": 12, "annot": [0, 1, 3, 7, 8, 28], "announc": 28, "anom": [], "anomali": [], "anonym": 18, "anoth": [0, 1, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 15, 22, 23, 25, 28, 29, 31], "ansatz": [0, 18, 28], "answer": [0, 1, 3, 5, 6, 19, 22, 23, 26, 28, 32], "antialias": [2, 6], "anticip": 4, "anymor": [1, 8], "anyon": [4, 8, 15], "anyth": [1, 15, 16, 25], "anytim": [26, 28], "anywai": [], "apach": 1, "apart": [11, 13, 30, 31], "api": [1, 21, 28], "appar": 2, "appear": [0, 1, 3, 13, 22, 25], "append": [1, 3, 4, 8, 9, 13, 19, 28, 31], "appendix": 23, "appli": [0, 1, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 18, 23, 25, 27, 28, 29, 31, 32], "applic": [0, 1, 3, 4, 5, 6, 7, 9, 12, 13, 16, 22, 25, 27, 28, 29, 30, 31, 32], "apply_gradi": 4, "approach": [1, 2, 4, 5, 6, 9, 10, 11, 12, 13, 15, 16, 18, 21, 23, 25, 27, 29, 30], "approch": 23, "appropri": [2, 6, 9, 12, 13, 17, 21, 25, 31, 32], "approv": 28, "approx": [0, 2, 3, 6, 10, 11, 13, 18, 23, 25, 28, 30, 31, 32], "approxim": [0, 1, 2, 3, 4, 5, 6, 7, 10, 11, 13, 19, 23, 25, 28, 29, 30, 31, 32], "apt": [0, 21, 23, 28], "aq": 25, "ar": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32], "aragorn": 28, "arang": [1, 3, 4, 6, 7, 9, 10, 12, 13, 28, 31], "arbitrari": [1, 4, 6, 8, 12, 13, 25, 30, 32], "arbitrarili": [0, 1, 11, 28, 31], "arc": 6, "architectur": [3, 4, 12], "archiv": 23, "area": [0, 3, 6, 27, 28], "argmax": [1, 11], "argmin": [4, 10, 14], "argsort": 11, "argu": [1, 13], "arguement": 19, "argument": [0, 2, 3, 5, 11, 12, 13, 17, 28, 29, 31, 32], "aris": [0, 6, 12, 13, 25, 28, 30, 32], "arithmet": [0, 13, 22, 28], "arm": [6, 29, 31], "armadillo": 22, "armin": [], "around": [0, 1, 4, 5, 6, 11, 18, 23, 25, 28, 32], "arrai": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 12, 13, 14, 16, 18, 21, 23, 25, 29, 30, 31, 32], "arrang": [3, 28], "arraybox": 13, "arriv": [0, 6, 9, 11, 22, 25, 28, 32], "arrow": 12, "arrowprop": 8, "art": [0, 1, 21], "articl": [0, 3, 4, 6, 10, 19, 28, 29, 30, 31, 32], "artifici": [0, 2, 7, 12, 27, 28], "artificialneuron": 12, "arug": 13, "arxiv": [3, 4, 31], "asarrai": [0, 6, 9, 29, 31], "asid": 29, "ask": [5, 6, 11, 12, 15, 19, 23, 32], "aspect": [0, 6, 21, 28, 29], "assembl": 3, "assembli": [0, 28], "assert": 4, "assess": [0, 6, 23, 28, 29, 32], "asset": [], "assici": 4, "assign": [0, 7, 8, 9, 12, 13, 14, 15, 24, 26, 27, 28], "associ": [0, 6, 9, 12, 14, 25, 28, 32], "assum": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "assumpt": [0, 3, 5, 6, 9, 11, 25, 28, 29], "ast": [0, 5, 6, 28, 32], "astyp": [4, 9, 10], "asymmetri": [0, 28], "asymptot": [4, 6, 31, 32], "atom": [0, 28], "attain": 31, "attempt": [0, 4, 6, 7, 8, 10, 28, 29, 31], "attend": 28, "attent": [0, 22, 28], "attract": [0, 10, 28], "attribut": [0, 9, 28], "audi": [0, 28], "audio": [3, 4], "august": [28, 29], "aurelien": [0, 27, 28], "austfjel": 6, "auth": 15, "authent": 15, "author": [0, 1, 10, 25], "authour": 28, "auto": [9, 10, 25], "auto_exampl": [23, 29], "autocor": 25, "autocorrelation_tim": 25, "autocorrelform": 25, "autocovari": 25, "autoencod": [4, 21, 28], "autoencond": 21, "autograd": [21, 28], "autom": [0, 21, 27, 28], "automac": 22, "automag": 28, "automat": [0, 1, 2, 3, 4, 11, 16, 21, 22, 28], "automobil": 3, "autonom": 4, "avail": [0, 1, 4, 6, 10, 11, 21, 22, 23, 24, 26, 27, 28, 32], "avali": 20, "averag": [0, 1, 3, 6, 9, 10, 13, 14, 25, 26, 28, 29, 32], "avoid": [0, 4, 5, 6, 9, 11, 13, 18, 22, 29, 31, 32], "awai": [2, 3, 6, 29, 31], "awar": [2, 10], "award": [26, 28], "ax": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 20, 22, 23, 28, 32], "axes3d": [2, 6, 13, 30, 31], "axes_grid1": 6, "axhlin": 8, "axi": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18, 22, 25, 28, 29, 30, 31, 32], "axiom": 5, "axvlin": [4, 8], "axvspan": 4, "b": [0, 1, 3, 4, 5, 6, 8, 9, 10, 12, 13, 14, 15, 16, 17, 19, 20, 25, 26, 28, 29, 30, 31, 32], "b1": 8, "b19db4": [], "b1bac4": [], "b2": 8, "b3": 8, "b35900": [], "b89784": [], "b_": [0, 1, 22], "b_0": 0, "b_1": [0, 2, 12, 13, 31], "b_2": [0, 13], "b_5": [13, 31], "b_group": 9, "b_i": [0, 1, 2, 12, 28], "b_ia_": [0, 28], "b_ia_i": 0, "b_index": 9, "b_j": [1, 12], "b_k": [0, 1, 12, 13, 31], "b_m": 12, "b_score": 9, "b_valu": 9, "ba": 31, "babcock": 28, "bach": 31, "bachelor": [24, 26], "back": [0, 3, 4, 5, 6, 8, 9, 10, 15, 16, 22, 25, 28, 31], "backbon": 22, "backend": [1, 4], "background": [27, 28], "backpropag": [1, 31], "backslash": [], "backtrack": 9, "backup": 22, "backward": [1, 2, 4, 12, 22, 31], "bad": [6, 17, 29], "badli": 25, "bag": [9, 21, 28], "bag_clf": 10, "baggin": 28, "baggingboot": 10, "baggingclassifi": 10, "baggingtre": 10, "bailei": [], "balanc": [6, 31, 32], "ballpark": 18, "band": 22, "bandwidth": 22, "banner": [], "bar": [0, 6, 11, 23, 28], "barber": 27, "bare": [4, 10], "base": [0, 1, 3, 4, 5, 7, 8, 9, 10, 14, 15, 16, 17, 21, 25, 26, 27, 28, 29, 30], "basi": [5, 7, 8, 10, 11, 12, 13, 22, 29, 30], "basic": [6, 8, 12, 13, 14, 15, 21, 23, 25, 28, 32], "basin": 31, "batch": [3, 4, 11, 12, 13, 30], "batch_shap": 4, "batch_siz": [1, 3, 4], "batchnorm": 4, "bay": 7, "bayesian": [5, 21, 27, 28], "bbbbbb": [], "beauti": [], "becam": [], "becaus": [0, 1, 2, 3, 4, 5, 6, 8, 9, 12, 13, 14, 28, 29, 30, 31, 32], "becom": [0, 1, 2, 5, 6, 7, 9, 12, 13, 19, 25, 28, 29, 30, 31, 32], "been": [0, 1, 2, 3, 4, 5, 6, 11, 12, 13, 20, 21, 22, 23, 28, 29, 31, 32], "befor": [0, 1, 2, 3, 4, 5, 6, 7, 8, 12, 13, 14, 16, 17, 18, 19, 20, 22, 23, 25, 28, 29, 31, 32], "beforehand": [0, 25, 28], "began": [], "begin": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 14, 15, 22, 25, 26, 28, 29, 30, 31, 32], "behav": [1, 6, 13, 30, 32], "behavior": [0, 1, 13, 28, 30, 31], "behaviour": [12, 31], "behind": [0, 1, 6, 8, 13, 28, 30], "being": [0, 1, 2, 3, 4, 5, 7, 8, 10, 11, 12, 13, 17, 20, 25, 28, 29, 30, 31], "believ": [9, 22], "belong": [7, 8, 9, 13, 14, 30], "below": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 18, 22, 23, 25, 28, 29, 30, 31, 32], "benchmark": 10, "benefici": [1, 13], "benefit": [0, 1, 4, 11, 13, 21, 28, 30, 31], "bengio": [1, 27, 28, 29, 31], "benign": [1, 7], "besid": [4, 5, 30], "bessel": [5, 29, 32], "best": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 15, 16, 18, 26, 28, 29, 30, 31, 32], "beta": [1, 3, 10, 11, 13, 16, 17, 19, 28, 29, 30], "beta1": [], "beta2": [], "beta_": [3, 13, 17, 29], "beta_0": [1, 3, 13, 29], "beta_1": [1, 3, 10, 13, 29, 31], "beta_1m_": 31, "beta_1x_i": 13, "beta_2": [3, 13, 31], "beta_2v_": 31, "beta_3": 3, "beta_i": [3, 31], "beta_j": [13, 29], "beta_k": 13, "beta_linreg": 13, "beta_m": 10, "beta_mg_m": 10, "beta_n": 3, "better": [0, 1, 2, 3, 4, 6, 9, 10, 11, 12, 13, 19, 20, 28, 29, 31, 32], "between": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 14, 15, 16, 17, 18, 19, 23, 25, 28, 29, 30, 31, 32], "beyond": [0, 1, 5, 6, 8, 13, 28, 29, 30, 31], "bf": [13, 14, 22, 25, 30], "bf5400": [], "bg": 28, "bgd": [13, 31], "bia": [0, 1, 2, 3, 5, 8, 9, 10, 12, 13, 20, 28, 29, 30], "bias": [1, 2, 3, 5, 6, 9, 12, 19, 31, 32], "bib": [], "bibliographi": 23, "bibtex": [], "big": [0, 1, 2, 5, 6, 14, 19, 31, 32], "bigger": [1, 6, 29], "bigr": 12, "bike": 9, "bilbo": 28, "billion": [3, 12, 21, 31], "bin": [7, 25], "binari": [0, 3, 5, 7, 9, 10, 12, 28], "binarycrossentropi": 4, "bind": 0, "binomi": [21, 25, 28], "binsboot": [6, 32], "bioinformat": 0, "biolog": [1, 12], "bios1100": [21, 28], "bird": [0, 3], "birth": 28, "bishop": [27, 28], "bit": [1, 4, 19, 22, 25, 28], "bitwis": 25, "bivari": 2, "bk": [13, 31], "bla": [22, 28], "black": [8, 9, 14], "blame": [], "block": [6, 10, 21, 22, 25, 28, 32], "blockquot": [], "blog": 28, "blogpost": 4, "blue": [0, 3], "bm": [], "bmatrix": [0, 1, 3, 5, 7, 8, 11, 13, 22, 28, 29, 30, 31], "bmi": 1, "bodi": [0, 1, 4, 12], "bold": 1, "boldfac": [0, 5, 16, 29, 30], "boldsymbol": [0, 1, 2, 3, 5, 6, 7, 8, 10, 11, 13, 14, 16, 17, 19, 23, 28, 30, 31], "boltzmann": [12, 21, 28], "book": [17, 23, 27, 28, 29, 32], "book1": 27, "bool": [], "boolean": [4, 17], "boost": [1, 9, 21, 28], "boostrap": 10, "bootstrap": [1, 13, 19, 21, 23, 28, 31], "born": 31, "borrow": 28, "boston_dataset": [], "bot": 8, "both": [0, 1, 4, 5, 6, 8, 9, 10, 13, 14, 15, 16, 17, 19, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "bottl": 7, "bottou": 31, "bound": [8, 12, 31], "boundari": [2, 4, 8, 11, 12], "bousquet": 31, "bower": [], "box": [4, 9], "boyd": [8, 13, 30], "bracket": [4, 25], "brain": [1, 7, 12], "branch": [9, 28], "break": [0, 4, 6, 11, 14, 28, 31], "breast": [5, 7, 11], "breviti": 13, "brew": [0, 21, 23, 28], "brg": 8, "brian": [], "brief": [23, 29], "briefli": [0, 16, 19, 28, 32], "bring": [0, 5, 6, 10, 29, 31], "britt": [26, 28], "broad": 0, "broadli": 28, "brought": [13, 21, 28], "brownle": 4, "browser": [15, 28], "brute": [3, 5, 11, 29], "bsd": [], "budget": 31, "buffer_s": 4, "bug": [], "bugfix": [], "bui": 4, "build": [0, 4, 5, 6, 10, 16, 22, 25, 28, 32], "built": [1, 3, 4, 6, 32], "bunch": 11, "bundl": [], "busi": [], "byte": [22, 28], "c": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 19, 20, 21, 22, 24, 25, 26, 27, 29, 30, 31, 32], "c1": [8, 11], "c2": [8, 11], "c4a2f5": [], "c5e478": [], "c9d1d9": [], "c_": [8, 9, 10, 13, 25, 30, 31], "c_0": 25, "c_1": 12, "c_2": 12, "c_3": 12, "c_4": 12, "c_i": [12, 13, 31], "c_k": 25, "ca": [1, 28], "caab6d": [], "cach": 10, "cal": [0, 8, 10, 12, 13, 30, 31], "calcul": [0, 1, 2, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 16, 19, 22, 25, 28, 31, 32], "california": 23, "call": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 18, 19, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "calor": [0, 29], "caltech": [], "cambridg": [13, 27, 30], "can": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27, 29, 30], "cancel": [0, 13, 28, 29], "cancer": [5, 10], "cancerpd": 7, "candid": [8, 9, 10, 31], "cannot": [0, 1, 4, 5, 6, 7, 8, 9, 23, 25, 29, 30, 31], "canopi": [0, 21, 23, 28], "canva": [15, 16, 19, 20, 23, 28], "cap": 5, "capabl": [0, 1, 8, 13, 21, 28], "capac": [2, 26], "capita": [], "caption": [20, 23], "captur": [4, 11, 12, 28], "car": [3, 4], "card": [0, 7, 28], "cardin": 1, "care": [11, 15, 19, 31], "carefulli": [13, 31], "carlo": [0, 6, 21, 25, 27, 28, 32], "carri": [2, 6, 7, 23, 32], "cart": 10, "case": [0, 1, 2, 3, 4, 5, 6, 7, 11, 12, 13, 14, 15, 16, 21, 22, 23, 28, 32], "casella": 27, "cast": 1, "cat": [3, 4], "catch": 0, "categor": [0, 1, 3, 9, 11, 28], "categori": [0, 1, 3, 7, 10, 12, 14, 28], "categorical_crossentropi": [1, 3], "caus": [0, 5, 6, 25, 28, 29, 30, 31, 32], "causal": 0, "causat": [0, 28], "cax": 1, "cb": [6, 28], "cbar": 1, "cc": [0, 1, 5, 13, 28, 29, 30, 31], "cc398b": [], "ccbb44": [], "ccc": [5, 12, 30], "cdf": 25, "cdot": [0, 2, 6, 12, 13, 14, 22, 25, 28, 30, 31, 32], "celebr": [13, 30], "cell": 4, "center": [0, 1, 6, 7, 8, 9, 11, 14, 18, 23, 25, 28, 29, 31, 32], "central": [0, 3, 5, 6, 8, 16, 20, 22, 28, 29], "centroid": [14, 25], "centroid_differ": 14, "centuri": 3, "certain": [0, 3, 6, 7, 9, 25, 28, 29, 32], "certainti": 32, "cf": [], "cf222e": [], "cffi": [], "cg": 13, "cha": [], "chain": [0, 1, 13, 21, 25, 28], "challeng": 15, "chanc": [1, 5, 13, 25, 31], "chang": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 13, 14, 15, 16, 19, 22, 23, 25, 28, 29, 30, 31, 32], "changelog": [], "channel": 3, "chapter": [0, 6, 10, 11, 16, 17, 19, 22, 23, 27, 28, 29, 30, 31, 32], "chapter3": [0, 23], "charact": [0, 3, 5, 28, 29, 30], "character": [8, 9, 10, 12, 25], "characterist": [0, 1, 3, 10, 13, 28], "charg": [0, 28], "charl": [], "charset": [], "chase": 4, "chatgpt": [15, 23], "chd": 7, "chddata": 7, "cheap": [5, 29, 30, 31], "cheaper": [1, 13, 31], "check": [1, 3, 4, 5, 11, 13, 15, 16, 19, 22, 28, 31], "checkmark": 3, "checkpoint": 4, "checkpoint_dir": 4, "checkpoint_prefix": 4, "chen": 10, "cheng": 29, "chiaramont": 2, "childcar": 16, "children": 16, "choic": [0, 1, 2, 3, 4, 6, 9, 12, 13, 14, 20, 22, 28, 29, 30, 31, 32], "choleski": [5, 22, 29, 30], "choos": [2, 3, 6, 9, 10, 11, 13, 14, 15, 18, 19, 23, 30, 32], "chosen": [0, 1, 2, 6, 8, 9, 10, 13, 16, 25, 28, 30, 31, 32], "chosen_datapoint": 1, "christian": 27, "christoph": [27, 28], "chunk": 31, "cifar": 3, "cifar10": 3, "circ": [1, 12, 31], "circl": [0, 8, 12, 29, 31], "circuit": 3, "circumfer": 9, "circumv": [1, 5, 13, 29, 30, 31], "citat": [], "cite": [20, 23], "ckpt": 4, "claim": [], "clariti": 25, "class": [0, 1, 3, 4, 6, 7, 8, 9, 11, 12, 13, 25, 28, 32], "class_nam": [3, 9], "class_val": 9, "class_valu": 9, "classic": [7, 9, 13], "classif": [0, 3, 5, 6, 7, 8, 11, 12, 21, 23, 27, 28, 29, 32], "classifi": [0, 1, 4, 7, 9, 10, 11, 28], "classificaton": 1, "classifii": 10, "claus": [], "clean": 1, "clear": [1, 5, 10, 12, 13, 31], "clearli": [0, 3, 5, 6, 7, 8, 25, 29, 30, 32], "clever": [1, 10], "clf": [0, 6, 8, 9, 10, 28, 29], "clf3": 0, "clf_lasso": 6, "clf_ridg": 6, "cli": 15, "click": [], "clip": [3, 25, 31], "clock": 31, "clone": [15, 26], "close": [0, 1, 2, 4, 6, 8, 9, 11, 12, 13, 14, 18, 25, 27, 28, 30, 31, 32], "closer": [3, 5, 13, 29, 30, 31], "closest": [8, 11, 13, 14], "closur": [21, 28], "cloud": [21, 28], "cluster": [0, 1, 4, 6, 11, 21, 28, 32], "cluster_label": 14, "cm": [1, 2, 3, 6, 8, 13, 30, 31], "cmap": [0, 1, 2, 3, 4, 6, 8, 9, 10, 28], "cmap_arg": 6, "cmd": [9, 15], "cn_": 25, "cnn": 12, "cnn_kera": 3, "cntk": [21, 28], "co": [0, 2, 3, 6, 9, 13, 28, 32], "code": [0, 3, 4, 6, 7, 8, 18, 19, 21, 22, 25, 27], "codec": [], "coef": [0, 28], "coef0": 8, "coef_": [0, 5, 6, 8, 9, 13, 16, 28, 29, 30, 31], "coeff": 5, "coeffici": [0, 3, 5, 6, 7, 8, 9, 13, 18, 22, 28, 29, 31, 32], "coerc": [0, 6, 28, 32], "coin": [10, 25], "coin_toss": 10, "col": [0, 11, 28, 29], "colab": [21, 28], "cold": 9, "colinear": [], "collabor": [20, 23], "collaps": 8, "collect": [2, 6, 10, 11, 17, 21, 25, 27, 28, 32], "collinear": [5, 29, 30], "color": [0, 3, 4, 6, 8, 9, 10, 25, 31], "color_channel": 3, "color_cod": 6, "colorbar": [1, 6, 20], "colsample_bytre": 10, "colsaobject": 10, "column": [0, 1, 2, 5, 6, 7, 8, 9, 11, 12, 16, 17, 18, 19, 22, 28, 29, 30, 31, 32], "columntransform": 9, "com": [4, 6, 15, 16, 19, 20, 21, 23, 27, 28, 30, 31, 32], "combin": [1, 2, 5, 6, 7, 10, 15, 18, 25, 32], "come": [0, 1, 3, 4, 5, 12, 13, 14, 15, 28, 29, 30, 31], "comfort": [], "command": [0, 1, 15], "comment": [0, 4, 5, 6, 20, 23], "commerci": [0, 21, 23, 28], "commit": 15, "commod": [0, 28], "common": [0, 1, 3, 5, 6, 7, 9, 11, 13, 14, 16, 23, 25, 28, 29, 30, 31, 32], "commonli": [0, 1, 4, 6, 7, 9, 13, 14, 29, 31, 32], "commonmark": [], "commun": [0, 12, 15, 23], "commut": 3, "commutatitav": 3, "compact": [0, 1, 3, 5, 6, 7, 9, 11, 12, 13, 14, 28, 29, 32], "compair": 0, "compar": [0, 3, 4, 5, 6, 11, 13, 18, 22, 23, 28, 29, 30, 31, 32], "comparison": [2, 4, 13], "compat": 7, "compens": 31, "compet": 0, "competit": 10, "compil": [0, 1, 3, 4, 13, 21, 22, 28], "complet": [0, 2, 3, 4, 9, 12, 15, 16, 17, 18, 19, 20, 28], "completenn": 12, "complex": [1, 5, 8, 9, 11, 12, 13, 16, 19, 28, 30, 31, 32], "complianc": [], "complic": [0, 1, 9, 13, 23, 28, 30, 31, 32], "compoment": 29, "compon": [0, 1, 3, 4, 5, 6, 7, 9, 14, 16, 21, 28, 29, 30, 32], "components_": 11, "compos": [9, 12, 13, 14, 21, 28], "compphys": [0, 6, 16, 20, 21, 23, 24, 26, 27, 28, 29, 30], "compress": [0, 28, 29], "compris": 6, "compromis": [5, 29, 30], "compulsori": [21, 28], "comput": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 15, 16, 17, 18, 21, 22, 23, 24, 25, 27, 28, 29, 30, 32], "computation": [0, 3, 6, 9, 13, 25, 28, 30, 31], "computationalscienceuio": 28, "computerlab": 23, "concaten": [2, 4, 6, 14], "concav": [1, 13, 29, 30], "concentr": 10, "concept": [0, 2, 21, 28, 29], "conceptu": [12, 13, 30], "concern": [0, 1, 4, 7, 28, 30], "concic": 28, "conclud": [0, 5, 13, 31], "conclus": 1, "cond": 2, "conda": [0, 1, 21, 23, 28], "condis": 29, "condit": [0, 2, 4, 5, 6, 8, 9, 11, 13, 25, 28, 29, 31, 32], "conduct": 21, "condwav": 2, "confid": [0, 5, 6, 7, 8, 19, 28, 29], "configur": 3, "confirm": [5, 12], "conform": [], "confus": [5, 6, 7, 10, 22, 29, 32], "confusion_matrix": 9, "congruenti": 25, "conjug": [4, 8], "conjugaci": 13, "conjunct": 3, "connect": [0, 1, 3, 4, 9, 11, 12, 13, 22, 28, 29, 30], "consensu": 31, "consequ": [5, 6, 8, 10, 12, 13, 29, 30, 31, 32], "consequenti": [], "conserv": [5, 14, 29, 30], "consid": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 16, 19, 22, 23, 25, 28, 29, 30, 31, 32], "consider": [0, 1, 5, 13, 28, 29, 30, 32], "consist": [1, 2, 3, 4, 6, 12, 13, 23, 25, 29, 30, 32], "consol": [], "const": [], "constant": [0, 2, 4, 5, 6, 8, 12, 13, 16, 18, 25, 28, 29, 30, 31], "constitu": [0, 28], "constitut": [2, 6, 32], "constrain": [1, 3, 5, 7, 11, 30], "constraint": [5, 6, 8, 13, 29, 30, 32], "construct": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 22, 25, 28, 29, 32], "constructor": [], "consum": 31, "contact": [0, 28], "contain": [0, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 18, 19, 22, 23, 25, 27, 28, 29, 30, 31, 32], "contemporari": 28, "content": [1, 15, 20, 21, 22, 28, 30, 31], "context": [6, 10, 13, 23, 30, 31, 32], "contigu": 22, "contin": 19, "continu": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 19, 22, 23, 25, 28, 29, 30, 31, 32], "contour": [9, 10, 13], "contourf": [8, 9, 10], "contract": [], "contrast": [1, 4, 9, 10, 12, 28, 31], "contribut": [0, 3, 5, 13, 18, 25, 28, 29, 30, 31], "contributor": [0, 23], "control": [0, 1, 3, 9, 13, 15, 21, 28], "conv": [3, 4], "conv2d": [3, 4], "conv2dtranspos": 4, "convei": 28, "conveni": [5, 6, 12, 13, 22, 23, 28, 30, 31, 32], "convent": [12, 29], "converg": [1, 2, 4, 5, 8, 13, 14, 18, 29, 30], "convergencewarn": [], "convers": [20, 31], "convert": [0, 1, 4, 5, 9, 11, 13, 22, 28, 29, 30], "converttomatrix": 4, "convex": [4, 5, 7, 29], "convinc": [13, 30], "convolut": [1, 4, 21, 28], "cool": [4, 9], "coolwarm": 6, "coordin": [5, 12, 14, 29, 30, 31], "coorel": [], "copi": [0, 1, 14, 15, 29], "copyright": [], "core": 10, "corel": 28, "coronari": 7, "corr": [5, 7, 11, 29], "correalt": [11, 21], "correct": [0, 1, 2, 3, 4, 5, 7, 13, 15, 19, 20, 22, 25, 28, 29, 30, 32], "correctli": [1, 2, 6, 7, 10, 18, 19, 23, 32], "correl": [0, 1, 3, 5, 6, 7, 10, 12, 13, 21, 25, 28, 30, 31, 32], "correlation_matrix": [5, 7, 11, 29], "correspond": [0, 3, 5, 6, 8, 9, 11, 12, 21, 22, 23, 25, 28, 29, 30, 32], "cortex": 12, "cosin": [3, 6, 32], "cost": [0, 2, 3, 5, 6, 7, 8, 9, 12, 13, 16, 17, 18, 19, 23, 28], "cost_deep_grad": 2, "cost_funct": 2, "cost_function_deep": 2, "cost_function_deep_grad": 2, "cost_function_grad": 2, "cost_grad": 2, "cost_histori": [], "cost_ol": [], "cost_ridg": [], "cost_sum": 2, "costli": 31, "costol": [13, 31], "could": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 17, 18, 22, 23, 25, 28, 29, 30, 31, 32], "coulomb": [0, 28], "count": [0, 9, 15, 24, 25, 26, 28], "counteract": 31, "counterpart": 28, "countor": 13, "coupl": [4, 5, 6, 32], "cours": [0, 1, 3, 5, 11, 15, 16, 17, 19, 20, 23, 26, 29, 32], "coursework": 15, "courvil": [27, 28, 29, 31], "cov": [5, 6, 11, 22, 25, 28, 29, 32], "cov_xi": [5, 11, 29], "cov_xx": [5, 11, 29], "cov_yi": [5, 11, 29], "covari": [0, 7, 21, 22, 28, 30], "covariance_matrix": [5, 11, 14], "cover": [0, 5, 21, 26, 27, 29, 30, 32], "covert": [0, 28], "covxi": 25, "covxx": 25, "covxz": 25, "covyi": 25, "covyz": 25, "covzz": 25, "cpu": 1, "craft": 3, "crash": 31, "creat": [1, 3, 4, 5, 9, 10, 11, 12, 15, 18, 19, 21, 28, 31], "create_biases_and_weight": 1, "create_convolutional_neural_network_kera": 3, "create_neural_network_kera": 1, "create_x": [5, 11], "creation": [], "credit": [0, 7, 26, 28], "crim": [], "crime": [], "criteria": [0, 4, 9, 10, 14, 25, 28], "criterion": [9, 10, 13, 18, 30, 31], "critic": [6, 23, 29], "critiqu": 23, "cross": [0, 1, 3, 7, 9, 10, 13, 15, 21, 25, 28, 29, 30, 31], "cross_entropi": 4, "cross_val_scor": [6, 32], "cross_valid": [7, 10], "crossvalid": [6, 32], "crucial": [1, 25, 31], "cs231": 3, "csr_matrix": [22, 28], "css": [], "csv": [0, 4, 6, 7, 9, 32], "ctnk": 1, "cubic": 0, "culprit": [], "cumbersom": [5, 32], "cumprod": [], "cumsum": [10, 11, 28], "cumul": [7, 10, 25, 31], "cumulative_heads_ratio": 10, "cup": 5, "current": [1, 2, 3, 4, 13, 14, 15, 16, 27, 30, 31], "curs": [0, 29], "curv": [6, 7, 10, 12, 23], "curvatur": [13, 30, 31], "custom": [6, 14], "custom_cmap": [9, 10], "custom_cmap2": [9, 10], "custom_lin": [], "cutpoint": 9, "cv": [6, 7, 10, 32], "cvxbook": [13, 30], "cvxopt": [5, 8, 29], "cycl": [1, 12], "cycler": [], "d": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 17, 19, 20, 22, 25, 26, 28, 29, 30, 31, 32], "d1": [], "d166a3": [], "d2": [], "d2_g_t": 2, "d2a8ff": [], "d4d0ab": [], "d71835": [], "d9dee3": [], "d_f": [13, 30], "d_g_t": 2, "d_net_out": 2, "da": 3, "dagger": [5, 22, 29, 30], "dai": [1, 9, 21], "damag": [], "damp": 3, "darget": 9, "darkr": 25, "dat": [0, 28], "dat_id": [0, 6, 7, 9, 28, 32], "data": [2, 4, 5, 8, 10, 12, 13, 14, 16, 19, 20, 22, 23, 27, 30, 31, 32], "data1": 14, "data2": 14, "data3": 14, "data4": 14, "data_id": [0, 6, 7, 9, 28, 32], "data_indic": 1, "data_panda": 28, "data_path": [0, 6, 7, 9, 28, 32], "databas": 1, "datafil": [0, 6, 7, 9, 28, 32], "datafram": [0, 4, 5, 7, 9, 11, 28, 29], "datapoint": [1, 5, 6, 7, 11, 13, 16, 30, 31, 32], "datasci": [15, 16, 19], "dataset": [0, 4, 6, 7, 8, 9, 10, 11, 13, 14, 16, 23, 28, 30, 31, 32], "datatyp": 4, "date": [15, 18, 23, 28, 29, 30, 31, 32], "daughter": 10, "davi": [], "david": 27, "davison": 32, "dbb7ff": [], "dbh": 1, "dbo": 1, "dcc6e0": [], "dcomposit": 22, "ddot": 2, "de": 31, "dead": 1, "deadlin": [15, 20], "deal": [0, 1, 3, 5, 6, 8, 11, 13, 14, 19, 22, 25, 28, 29, 30, 31], "dealt": 0, "debt": 7, "debug": [0, 5, 6, 29, 30, 31, 32], "debugg": [], "decad": [0, 3, 31], "decai": [0, 13, 25, 28], "decemb": [26, 28], "decent": 10, "decid": [0, 2, 3, 5, 6, 9, 18, 29, 30, 31, 32], "decim": [0, 28], "decis": [0, 1, 8, 11, 21, 27, 28], "decision_funct": 8, "decision_tre": 9, "decisiontreeclassifi": [9, 10], "decisiontreeregressor": [0, 9, 10], "declar": [0, 4, 20, 22, 28], "declare_namespac": [], "decompos": [5, 6, 22, 29, 30], "decomposit": [0, 6, 12, 28], "decompost": [5, 29, 30], "deconvolut": 3, "decorrel": [10, 13, 31], "decreas": [1, 2, 4, 5, 6, 10, 11, 13, 19, 30, 31, 32], "dedic": 20, "deduc": [0, 28], "deep": [3, 7, 12, 13, 21, 27, 29, 30], "deep_neural_network": 2, "deep_param": 2, "deep_tree_clf": [9, 10], "deep_tree_clf1": 9, "deep_tree_clf2": 9, "deepen": [5, 21, 28], "deeper": [0, 3, 4, 28], "deeplearningbook": [27, 28, 30, 31], "deer": 3, "def": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 16, 17, 25, 28, 29, 30, 31, 32], "def_covari": 25, "default": [0, 1, 2, 4, 6, 7, 22, 28, 29], "default_tim": 4, "defect": [5, 29, 30], "defici": [5, 29, 30], "defin": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 18, 19, 22, 23, 25, 29, 30, 31, 32], "definit": [1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 22, 25, 29, 30, 31, 32], "defint": 25, "degre": [3, 5, 6, 8, 9, 10, 11, 15, 16, 19, 20, 23, 25, 28, 30, 31, 32], "deisenroth": 29, "del": 1, "delet": [6, 15], "delimit": 4, "deliv": [15, 24, 28], "delta": [0, 2, 3, 6, 8, 12, 13, 14, 28, 31], "delta_": [1, 22], "delta_0": 3, "delta_1": 3, "delta_2": 3, "delta_3": 3, "delta_4": 3, "delta_5": 3, "delta_h": [0, 1, 28], "delta_j": [3, 12], "delta_k": 12, "delta_l": [1, 3], "delta_momentum": [13, 31], "delta_n": [0, 3, 28], "delug": 21, "delv": 0, "demand": [13, 30], "demonstr": [0, 3, 5, 6, 7, 11, 12, 19, 21, 28, 29, 30, 31, 32], "den": 4, "denomin": [1, 5, 31], "denot": [1, 2, 6, 7, 13, 25, 30, 31], "dens": [1, 3, 4], "densiti": [0, 2, 6, 25, 32], "depart": [26, 28, 29, 30, 31, 32], "depend": [0, 1, 2, 4, 5, 6, 7, 8, 11, 12, 13, 15, 16, 21, 22, 23, 25, 28, 29, 30, 31], "depict": 25, "deploy": [0, 21, 23, 28], "depth": [0, 3, 9, 10, 22, 32], "der": [], "deriv": [0, 1, 2, 6, 7, 8, 10, 11, 13, 18, 21, 23, 28], "derivati": 13, "derivative_fn": 13, "descend": [5, 9, 11, 29, 30], "descent": [0, 1, 3, 7, 8, 12, 28, 29], "describ": [0, 2, 4, 5, 6, 8, 10, 11, 12, 13, 19, 20, 22, 23, 28, 31, 32], "descript": [0, 8, 9, 20, 23, 28], "design": [0, 1, 3, 4, 5, 6, 7, 10, 11, 12, 13, 17, 18, 23, 28, 30, 31, 32], "designmatrix": [0, 28], "desir": [0, 2, 4, 5, 13, 14, 28, 29, 30, 31], "desktop": 15, "despit": [1, 12, 31], "destroi": 22, "det": [5, 22, 29, 30], "detail": [0, 6, 11, 13, 14, 18, 22, 23, 29, 30, 31], "detect": [3, 8, 12], "determin": [0, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 18, 22, 25, 28, 29, 30, 31, 32], "determinist": [7, 13, 25, 30, 31], "dev": [1, 23], "develop": [0, 3, 5, 8, 10, 11, 12, 21, 22, 23, 28, 29], "deviat": [0, 1, 2, 4, 5, 6, 17, 18, 19, 23, 25, 28, 29, 31, 32], "devis": 12, "df": [4, 8, 11, 13, 28], "df1": 28, "di": [], "diag": [5, 8, 29, 30, 31], "diagnost": [1, 10], "diagon": [0, 5, 7, 13, 18, 19, 22, 25, 28, 29, 30, 31], "diagonaliz": [5, 29, 30], "diagram": 10, "diagsvd": 6, "dice": [6, 25, 32], "dict": [6, 8], "dictionari": [], "did": [0, 1, 5, 6, 7, 10, 11, 14, 16, 23, 28, 32], "die": 1, "diff": 2, "diff1": 2, "diff2": 2, "diff_ag": 2, "diffeent": 8, "differ": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 32], "differenti": [0, 3, 16, 21, 22, 28, 29, 30], "difficult": [0, 1, 6, 10, 13, 25, 28, 31, 32], "difficulti": [0, 1, 13, 28, 30, 31], "diffonedim": 2, "digit": [0, 1, 3, 4, 6, 26, 28], "dilemma": [13, 31], "dilut": 1, "dim": [4, 11, 14, 22], "dimens": [0, 1, 2, 3, 4, 5, 8, 11, 14, 16, 22, 28, 29, 30], "dimension": [0, 4, 5, 6, 9, 11, 13, 14, 19, 21, 22, 23, 28, 29, 30, 31, 32], "dimensionless": [0, 3, 28], "diment": 22, "diminish": 31, "dimnsion": 4, "diod": 3, "direct": [0, 1, 2, 4, 11, 12, 13, 14, 28, 29, 30, 31], "directli": [1, 4, 5, 6, 18, 25, 29, 30], "directori": [], "disadvantag": [0, 28, 31], "disappear": [3, 6, 32], "disc_loss": 4, "disc_tap": 4, "discard": [6, 11, 31, 32], "disciplin": [0, 3, 12], "disclaim": 25, "discord": 28, "discourag": [13, 15, 30], "discov": [0, 28], "discover": 5, "discret": [1, 3, 5, 7, 13], "discrimin": [4, 7, 10, 11], "discriminator_loss": 4, "discriminator_loss_list": 4, "discriminator_model": 4, "discriminator_optim": 4, "discuss": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 19, 20, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "diseas": 7, "disguis": [6, 29, 31], "disk": 31, "disord": [1, 7], "displai": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 23, 25, 28, 29, 31, 32], "displaystyl": [0, 5, 17, 28, 29, 30, 31], "disregard": [0, 28], "dissimilar": [11, 14], "dist": 14, "distanc": [8, 9, 11, 14, 25], "distance_list": 9, "distinct": [3, 7, 8, 9, 10, 14], "distinctli": 8, "distinguish": [0, 4, 7, 8, 25, 28], "distplot": [], "distribut": [0, 1, 4, 6, 7, 10, 11, 13, 14, 18, 19, 21, 22, 23, 28, 29, 30, 31], "distrubut": [0, 21, 23, 28], "div": [], "dive": [0, 8, 22, 28], "diverg": [1, 13, 30, 31], "divid": [0, 1, 3, 5, 6, 7, 8, 9, 11, 12, 18, 19, 25, 28, 29, 31, 32], "divis": [6, 8, 9, 13, 18, 22, 25, 31, 32], "dl": [], "dm": [], "dna": 7, "dnn": [0, 1, 2, 4, 12, 28], "dnn1": 4, "dnn2_gru2": 4, "dnn_kera": 1, "dnn_model": 1, "dnn_numpi": 1, "dnn_scikit": [0, 1, 28], "do": [0, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 22, 23, 28, 29, 30, 32], "doc": [0, 15, 16, 19, 21, 23, 24, 26, 27, 28], "document": [4, 13, 15], "docutil": [], "doe": [0, 1, 2, 3, 4, 5, 6, 8, 10, 11, 12, 13, 15, 16, 17, 18, 19, 22, 23, 25, 28, 31, 32], "doesn": [3, 9, 12, 28, 31], "dog": [1, 3, 4], "dollar": [], "domain": [5, 8, 13, 23, 30, 32], "domcontentload": [], "domin": [0, 28], "don": [0, 1, 3, 5, 6, 8, 11, 13, 15, 16, 21, 23, 28, 29, 31], "done": [0, 2, 3, 4, 5, 6, 9, 10, 11, 13, 16, 20, 22, 23, 28, 29, 30, 31, 32], "dot": [0, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 18, 22, 23, 25, 28, 29, 30, 31, 32], "doubl": [3, 4, 16, 22, 28], "doubli": 1, "doubt": 23, "down": [0, 3, 6, 9, 11, 12, 13, 30, 31], "download": [0, 1, 3, 5, 6, 15, 20, 22, 27, 28], "downsampl": 3, "dozen": 1, "dq": [6, 32], "draft": 20, "drag": 13, "dragon": [], "dramat": 11, "drastic": 4, "draw": [4, 6, 10, 13, 30, 32], "drawback": [0, 1, 3, 13, 29, 30, 31], "drawn": [1, 4, 6, 7, 11, 25, 28, 32], "drive": [3, 4], "driven": 3, "drop": [0, 1, 5, 6, 11, 13, 25, 28, 29, 30, 32], "dropna": [0, 6, 28, 32], "dropout": 4, "dt": [2, 3, 13, 25], "dtype": [0, 1, 3, 4, 14, 22, 28], "dual": [], "dub": [0, 28], "duboi": [], "due": [1, 2, 5, 6, 8, 10, 12, 13, 18, 26, 28, 29, 30, 31, 32], "dugard": [], "dummi": [], "dure": [0, 1, 3, 4, 8, 9, 11, 20, 21, 23, 28, 31, 32], "dwell": [], "dwh": 1, "dwo": 1, "dx": [2, 3, 8, 25], "dx_1": 25, "dx_1p": [6, 32], "dx_2p": [6, 32], "dx_mp": [6, 32], "dx_n": 25, "dxp": [6, 32], "dy": [1, 8, 25], "dynam": 4, "dz": 8, "e": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 25, 26, 28, 29, 30, 31, 32], "e1e1e1": [], "e_": [0, 2, 28], "each": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 21, 22, 24, 25, 26, 28, 29, 30, 31, 32], "eager": 32, "eapprox": [0, 28], "earli": [1, 13, 31], "earlier": [0, 5, 7, 8, 9, 11, 12, 13, 20, 28, 29], "earthexplor": 6, "eas": [6, 9, 14, 32], "easi": [0, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 21, 22, 28, 29, 30, 31, 32], "easier": [5, 6, 8, 9, 13, 15, 20, 23, 25, 28, 29, 30, 32], "easiest": [13, 18], "easili": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 22, 23, 28, 29, 30, 31, 32], "eastern": [26, 28], "ebind": [0, 28], "eblock": 9, "ec8e2c": [], "econom": [], "econometr": 28, "economi": 5, "ecosystem": [21, 28], "ect": 24, "edg": 3, "edgecolor": [6, 32], "edit": [], "editor": [15, 20], "edu": [13, 23, 30], "educ": [0, 23, 28, 32], "ee6677": [], "eff": 25, "effect": [1, 4, 10, 13, 16, 17, 18, 25, 31], "effic": 1, "effici": [0, 3, 10, 13, 21, 22, 25, 28, 31], "effort": 19, "efron": [6, 32], "egrad": 13, "eig": [5, 11, 13, 22, 25, 28, 29, 30, 31], "eigen": 25, "eigenpair": [5, 11, 29, 30], "eigenvalu": [0, 5, 8, 11, 13, 22, 28, 29, 30, 31], "eigenvector": [5, 11, 13, 29, 30], "eight": [22, 28], "eigval": [22, 25, 28], "eigvalu": [11, 13, 30, 31], "eigvec": [22, 25, 28], "eigvector": [11, 13, 30, 31], "eir": [26, 28], "eispack": [22, 28], "either": [1, 5, 6, 7, 8, 9, 10, 11, 13, 18, 19, 23, 25, 28, 29, 30, 31, 32], "eivind": 26, "eivinsto": 26, "ekstr\u00f8m": 4, "elabor": 25, "elarn": 3, "electr": [0, 3, 12, 28], "electron": 28, "eleg": 11, "element": [1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 19, 20, 21, 22, 23, 27, 29, 31, 32], "elementari": [10, 13, 22], "elementwis": [3, 13], "elementwise_grad": [2, 13], "elessar": 28, "elif": 14, "elim": 22, "elimin": [3, 8], "elin": [26, 28], "ell_": [], "ellipsi": 16, "els": [1, 3, 4, 7, 9, 12, 13, 16, 22], "elu": 1, "elus": [0, 28], "em": [], "email": [20, 24, 26, 28], "emb": [], "embed": [0, 11, 29], "embodi": [6, 23, 32], "emit": 25, "emner": 27, "emph": 31, "emphas": [0, 10, 21, 28], "emphasi": [0, 21, 27, 28], "empir": [1, 11, 25], "emploi": [0, 1, 5, 6, 11, 13, 25, 28, 29, 30, 32], "employ": 0, "empti": [6, 10, 15, 32], "emul": 12, "en": [21, 23, 27], "enabl": [11, 31], "enbodi": [6, 32], "encod": [0, 3, 5, 9, 11, 14, 28, 29, 30], "encompass": [0, 23, 25], "encount": [0, 1, 5, 7, 13, 15, 23, 25, 28, 29, 30, 31], "encourag": [15, 23], "end": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 20, 22, 25, 26, 28, 29, 30, 31, 32], "endblock": [], "endfor": [], "endif": [], "endors": [], "endpoint": [3, 6], "energi": [0, 4, 6, 32], "enforc": 12, "eng": 27, "engin": [0, 1, 3, 4, 21, 28], "english": 23, "enjoi": 31, "enocurag": 23, "enorm": 3, "enough": [0, 6, 13, 28, 30, 31, 32], "ensembl": [1, 9, 28], "ensur": [0, 1, 2, 3, 5, 6, 11, 13, 18, 25, 29, 30, 31, 32], "entail": 28, "enter": [5, 6, 29, 30, 31], "enthought": [0, 21, 23, 28], "entir": [1, 3, 7, 9, 21, 25, 28, 31], "entireti": [], "entiti": [9, 12, 22, 28], "entri": [0, 5, 8, 11, 12, 22, 28, 29, 31, 32], "entropi": [1, 3, 7, 10, 13, 28, 30, 31], "enumer": [0, 1, 2, 3, 4, 6, 8, 28, 29, 31], "env": 25, "environ": [2, 21, 23, 28], "environemnt": 15, "eo": [0, 6, 32], "eol": 0, "eosfit": 0, "epoch": [0, 1, 3, 4, 12, 13, 28, 31], "eppstein": [], "epsilon": [0, 5, 6, 7, 13, 23, 28, 29, 30, 31, 32], "epsilon_": [0, 28], "epsilon_0": [0, 28], "epsilon_1": [0, 28], "epsilon_2": [0, 28], "epsilon_i": [0, 28, 29], "eq": [3, 13, 14, 22, 25, 30], "eqnarrai": [3, 5, 6, 32], "equal": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 13, 14, 16, 18, 22, 23, 25, 28, 29, 30, 31, 32], "equat": [1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 17, 19, 22, 25, 28, 31, 32], "equilibrium": [2, 12], "equiv": [3, 13, 22, 25, 30, 31], "equival": [0, 1, 5, 7, 8, 11, 13, 21, 22, 28, 29, 30, 31, 32], "equivel": 19, "eras": [], "erf": 25, "eriador": 28, "eric": [], "err": [0, 10], "err_": [6, 32], "err_sqr": 2, "errat": [13, 30, 31], "erron": 2, "error": [1, 2, 4, 5, 6, 7, 9, 11, 12, 13, 15, 16, 17, 18, 19, 21, 22, 23, 25, 31], "error_estimate_corr_tim": 25, "error_hidden": 1, "error_output": 1, "escap": [13, 30, 31], "escapehtml": [], "especi": [1, 3, 9, 12, 13, 15, 18, 23, 31], "essenti": [0, 5, 6, 9, 10, 12, 14, 15, 23, 25, 29, 30, 31], "establish": [0, 6, 10, 11, 16, 23], "estim": [0, 1, 5, 6, 7, 10, 11, 13, 19, 21, 25, 28, 29, 30, 31], "estimated_mse_fold": [6, 32], "estimated_mse_kfold": [6, 32], "estimated_mse_sklearn": [6, 32], "et": [0, 2, 4, 16, 17, 20, 27, 28, 29, 30, 32], "eta": [0, 1, 3, 8, 12, 13, 18, 28, 30, 31], "eta0": [8, 13], "eta_": 13, "eta_j": 31, "eta_t": [13, 31], "eta_v": [0, 1, 3, 28], "etc": [0, 1, 3, 5, 7, 8, 9, 11, 12, 13, 14, 21, 22, 23, 25, 29, 30, 31], "ethic": 21, "etsim": 32, "euclidean": [0, 14, 29, 31], "euler": [], "evalu": [0, 2, 3, 4, 5, 6, 9, 13, 15, 16, 17, 19, 23, 25, 28, 29, 30, 31, 32], "evalut": [13, 23], "even": [0, 1, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 21, 22, 25, 28, 29, 30, 31, 32], "evenli": 4, "event": [5, 7, 10, 25, 32], "eventu": [0, 5, 6, 11, 12, 13, 23, 26, 29, 30, 31, 32], "everi": [0, 1, 2, 3, 4, 5, 6, 9, 10, 11, 12, 13, 14, 15, 21, 25, 26, 28, 29, 30, 31, 32], "everyth": [4, 12, 16, 18], "everywher": [4, 13, 30], "evolv": 0, "exact": [0, 5, 11, 12, 13, 22, 25, 28, 29, 31], "exactli": [0, 3, 4, 6, 12, 18, 21, 29, 31, 32], "exam": 28, "examin": [6, 32], "exampl": [0, 5, 11, 12, 13, 15, 16, 18, 20, 21, 22, 23, 25, 27], "exce": [1, 12, 13, 31], "exceed": 31, "excel": [0, 1, 4, 5, 10, 20, 23, 28, 29], "except": [3, 4, 6, 8, 9, 22], "excess": [0, 28], "exchang": 31, "excit": 0, "exclud": [1, 6, 12, 29, 31, 32], "exclus": [0, 1, 3, 6, 25, 28, 32], "execut": [2, 5, 13, 15, 29, 30, 31], "exemplari": [], "exemplifi": [13, 31], "exercic": [26, 28], "exercis": [5, 21, 23, 24, 26, 28, 30, 31, 32], "exhaust": [6, 31, 32], "exhibit": [0, 5, 6, 8, 28, 29, 32], "exist": [0, 1, 2, 3, 5, 6, 7, 8, 9, 13, 19, 22, 23, 28, 30, 31, 32], "exit": [5, 22, 29, 30], "exp": [0, 1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 16, 17, 19, 25, 29, 30, 31, 32], "exp_term": 1, "expand": [5, 7, 11, 13, 30], "expans": [0, 3, 5, 8, 10, 12, 13, 28, 29, 30], "expect": [0, 1, 5, 6, 7, 11, 12, 13, 15, 18, 21, 23, 28, 29, 31], "expectation_value_of_h_wrt_p": 25, "expens": [6, 10, 13, 16, 30, 31], "experi": [0, 1, 6, 8, 13, 15, 21, 23, 28, 29, 30, 31, 32], "experiment": [0, 4, 6, 9, 25, 28, 32], "expert": [1, 9], "explain": [0, 6, 9, 10, 11, 13, 16, 19, 23, 28, 30], "explained_variance_ratio_": 11, "explan": [], "explanatori": [0, 28], "explicit": [0, 3, 6, 13, 22, 23, 28, 29, 30, 31], "explicitli": [0, 4], "explod": 1, "exploit": [0, 3, 12, 13, 28, 31], "explor": [1, 4, 6, 8, 13, 18, 21, 23, 28, 30, 31], "expon": 1, "exponenti": [0, 1, 5, 6, 10, 13, 25, 28, 30], "export": [9, 15, 16, 19, 20], "export_graphviz": 9, "export_text": 9, "exporttext": 9, "expos": 21, "express": [0, 2, 3, 5, 6, 7, 10, 12, 13, 18, 22, 23, 25, 28, 30, 31, 32], "exptmean": 25, "exptvari": 25, "extend": [0, 2, 7, 11, 13, 21, 28, 31], "extend_path": [], "extens": [0, 12, 15, 21, 28], "extent": [0, 1, 6, 27, 32], "extern": [3, 6, 9], "extra": [1, 3, 5, 15, 26, 28, 29, 30], "extract": [0, 3, 5, 6, 7, 8, 11, 13, 16, 17, 22, 28, 29], "extrapol": [0, 28], "extrem": [0, 1, 4, 5, 6, 7, 8, 9, 13, 15, 16, 22, 29, 30, 31], "extremum": [13, 30], "extrins": 11, "ey": [0, 5, 6, 13, 14, 18, 22, 28, 29, 30, 31], "f": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 12, 13, 14, 15, 16, 17, 18, 19, 22, 25, 26, 28, 29, 30, 31, 32], "f1": 13, "f11": [0, 28], "f12": [0, 28], "f13": [0, 28], "f1_grad": 13, "f1d": 13, "f2": 13, "f26196": [], "f2_grad_x1": 13, "f2_grad_x1_analyt": 13, "f2_grad_x2": 13, "f2_grad_x2_analyt": 13, "f2f2f2": [], "f3": 13, "f3_grad": 13, "f3_grad_analyt": 13, "f4": 13, "f4_grad": 13, "f4_grad_analyt": 13, "f5": 13, "f5_grad": 13, "f5a394": [], "f5ab35": [], "f5f5f5": [], "f6": 13, "f6_for": 13, "f6_for_grad": 13, "f6_grad_analyt": 13, "f6_while": 13, "f6_while_grad": 13, "f7": 13, "f78c6c": [], "f7_grad": 13, "f7_grad_analyt": 13, "f8": 13, "f8_grad": 13, "f8f8f2": [], "f9": [0, 13, 28], "f9_altern": 13, "f9_alternative_grad": 13, "f9_grad": 13, "f_": 10, "f_0": [3, 10], "f_1": [10, 13, 30], "f_2": [12, 13, 30], "f_3": 12, "f_d": 25, "f_grad": 13, "f_grad_analyt": 13, "f_i": [0, 6, 12, 16, 32], "f_m": [3, 10], "f_n": 3, "f_vec": 2, "face": [13, 28, 30], "facecolor": [6, 8, 25, 32], "facil": [0, 21], "facilit": 12, "fact": [0, 1, 3, 5, 9, 11, 12, 13, 28, 29, 30, 31], "facto": 31, "factor": [0, 1, 3, 5, 6, 9, 10, 11, 13, 22, 25, 28, 29, 30], "factori": 13, "fad000": [], "fade": 6, "fae4c2": [], "fafab0": [9, 10], "fail": [0, 6, 13, 26, 28, 30, 32], "failur": 7, "fairli": [1, 2, 18, 25, 31], "faisal": [16, 29], "fake": 4, "fake_loss": 4, "fake_output": 4, "fall": [8, 9, 24], "fals": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 14, 16, 17, 22, 28, 29, 30, 31, 32], "famili": [0, 7, 8, 25, 29, 31], "familiar": [0, 3, 5, 6, 8, 15, 21, 22, 23, 25, 28, 32], "famou": [6, 12], "far": [0, 3, 4, 5, 6, 8, 11, 12, 13, 14, 16, 20, 28, 29, 30, 31], "fashion": [0, 9, 10, 28, 31], "fast": [1, 3, 6, 10, 12, 13, 21, 25, 28, 30, 31, 32], "faster": [1, 11, 13, 31], "fastest": [13, 22, 30], "fatal": [], "favor": [7, 31], "favorit": 25, "fc": 3, "fcfcfc": [], "fdac54": [], "fdf2e2": [], "featur": [0, 1, 3, 5, 6, 7, 8, 10, 11, 12, 13, 15, 17, 18, 19, 21, 25, 28, 30, 31, 32], "feature_nam": [1, 7, 9], "feautur": 9, "fed": 1, "feed": [0, 2, 3, 11, 21, 28], "feed_forward": 1, "feed_forward_out": 1, "feed_forward_train": 1, "feedback": [4, 20, 28], "feeddorward": 4, "feedforward": [1, 4, 12], "feel": [0, 5, 6, 11, 13, 15, 16, 18, 21, 23, 26, 28], "feet": [], "fefef": [], "fefeff": [], "felt": 23, "fenc": [], "fernando": [], "fetch": [6, 15], "few": [1, 3, 4, 5, 9, 17, 18, 19, 25, 28], "fewer": [0, 9, 11, 19, 28, 31], "ff7b72": [], "ff9492": [], "ffa07a": [], "ffa657": [], "ffb757": [], "ffd700": [], "ffd900": [], "ffd9002e": [], "ffffff": [], "ffnn": [1, 12], "fi": [], "field": [0, 3, 6, 12, 19, 21], "fieldmask": [], "fifth": [0, 6, 28], "fig": [0, 1, 2, 3, 4, 6, 7, 12, 13, 14, 23, 28], "fig_id": [0, 6, 7, 9, 28, 32], "figaxi": 25, "figsiz": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 28, 32], "figur": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 16, 21, 23, 28, 29, 30, 31, 32], "figure_id": [0, 6, 7, 9, 28, 32], "figurefil": [0, 6, 7, 9, 28, 32], "file": [0, 4, 5, 6, 7, 9, 15, 20, 23, 28, 32], "file_prefix": 4, "filenam": 28, "fill": [5, 9, 18, 29, 30], "fill_valu": [], "filter": [3, 4], "final": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 18, 20, 23, 24, 25, 26, 28, 30, 32], "financ": 0, "find": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 21, 23, 25, 28, 29, 30, 31], "fine": [0, 14], "finish": [2, 20], "finit": [3, 5, 6, 12, 13, 17, 25, 29, 30, 32], "finnicki": 15, "fire": [], "first": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 18, 19, 22, 23, 25, 26, 27, 29, 31, 32], "first_moment": 31, "first_term": 31, "firsteigvector": 11, "fit": [1, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 17, 18, 19, 23, 25, 29, 31, 32], "fit_beta": 29, "fit_intercept": [0, 5, 6, 16, 29, 30, 31, 32], "fit_mod": 9, "fit_theta": [6, 31], "fit_transform": [0, 6, 8, 9, 11, 15, 19, 32], "fiti": [0, 28], "five": [0, 9, 28, 29], "fix": [0, 3, 4, 6, 10, 11, 12, 13, 23, 28, 32], "flag": 4, "flat": [12, 13, 30, 31], "flatten": [1, 3, 4, 5, 22], "flavor": [], "flexibl": [1, 6, 8, 10, 12, 28, 31, 32], "flip": [26, 28], "float": [0, 3, 4, 5, 9, 11, 13, 14, 22, 28, 29, 30], "float32": [4, 9], "float64": [4, 22, 28], "flop": [5, 22, 29, 30], "flow": [1, 4, 12], "fluctuat": [5, 31], "fly": 11, "fm": 0, "fmax": 3, "fmesh": 13, "fn": 7, "focu": [0, 3, 4, 5, 6, 15, 21, 23, 27, 28, 29, 30, 31, 32], "focus": [1, 6, 7, 22, 29, 31], "fold": [6, 9, 23], "folder": [0, 4, 6, 15, 20, 23, 28], "follow": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 19, 20, 21, 22, 23, 25, 26, 27, 28, 29, 30, 31, 32], "font": [7, 20, 25, 28], "fontdict": 25, "fontsiz": [1, 6, 8, 9, 10, 25], "fontweight": 1, "footprint": [3, 31], "foral": [8, 29], "forc": [0, 5, 6, 10, 11, 29, 30, 31], "forcast": 4, "forcier": [], "forecast": [4, 12], "forest": [0, 1, 9, 21, 28], "forget": [11, 31], "form": [0, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 16, 21, 22, 23, 25, 28, 29, 30, 31, 32], "formal": [3, 4, 14, 18, 25], "format": [0, 1, 3, 4, 6, 7, 8, 9, 10, 11, 20, 21, 25, 27, 32], "format_data": 4, "formatstrformatt": [6, 13, 30, 31], "formatt": [], "formul": [4, 6, 11, 14], "formula": [3, 13, 25, 30], "forth": [4, 12], "fortran": [0, 21, 22, 28], "fortran2003": [21, 28], "fortran2008": 23, "fortran90": 25, "fortun": [0, 11, 29], "forward": [0, 3, 6, 21, 22, 28, 31, 32], "found": [1, 2, 4, 5, 6, 12, 13, 19, 20, 23, 28, 29, 31, 32], "foundat": [21, 28], "four": [4, 5, 6, 8, 12, 22, 24, 26, 28, 30], "fourier": [0, 28], "fourierdef1": 3, "fourierdef2": 3, "fourierseriessign": 3, "fourth": [12, 28, 29], "fp": 7, "frac": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "fraction": 9, "frame": [7, 31], "framework": [1, 8, 10, 25], "frank": [5, 11], "frankefunct": [5, 6, 11], "fredli": [26, 28], "free": [0, 6, 11, 13, 15, 16, 18, 21, 22, 23, 25, 26, 27, 28], "freecodecamp": 21, "freedom": [5, 30], "freeli": [0, 23], "freez": 15, "frequenc": [3, 6, 7, 25, 32], "frequent": [0, 8, 9, 13, 30], "frequentist": 21, "fresh": 10, "fridai": [15, 26, 28], "friedman": [6, 19, 23, 27, 28], "friendli": 4, "fro": 23, "frodo": 28, "frog": 3, "from": [0, 1, 2, 3, 4, 6, 7, 8, 9, 11, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27], "from_cod": 9, "from_logit": [3, 4], "from_tensor_slic": 4, "front": [0, 4, 5, 28, 29, 30], "frustrat": 15, "fulfil": [2, 5, 12, 29, 30], "full": [1, 3, 5, 7, 9, 10, 13, 25, 28, 29, 30], "full_matric": [5, 29, 30], "fulli": [3, 6, 12, 25, 32], "fullnam": [], "fun": [21, 28], "func": 2, "function": [2, 3, 4, 5, 9, 14, 15, 16, 17, 18, 19, 20, 21, 22], "functionali": 11, "fundament": [0, 6, 21, 28, 32], "funtion": 2, "furnish": [], "further": [2, 7, 9, 19, 28], "furthermor": [0, 3, 5, 6, 7, 11, 12, 13, 21, 23, 28, 29, 30, 31, 32], "futur": [0, 4, 8, 9, 28], "fy": [15, 23, 24, 26, 27, 28], "fys4155": 23, "fys5419": [27, 28], "fys5429": [27, 28], "f\u00f8470": [26, 28], "g": [0, 1, 2, 3, 4, 6, 8, 9, 10, 11, 13, 15, 18, 19, 25, 28, 29, 30, 31, 32], "g0": 2, "g_": [2, 9, 10, 31], "g_0": 2, "g_1": [2, 10], "g_2": [2, 10], "g_analyt": 2, "g_dnn_ag": 2, "g_euler": 2, "g_i": 2, "g_m": [3, 10], "g_n": 3, "g_re": 2, "g_t": [2, 31], "g_t_d2t": 2, "g_t_d2x": 2, "g_t_dt": 2, "g_t_hessian": 2, "g_t_hessian_func": 2, "g_t_jacobian": 2, "g_t_jacobian_func": 2, "g_trial": 2, "g_trial_deep": 2, "g_vec": 2, "gain": [1, 5, 7, 9, 10, 13, 29, 30], "galleri": [0, 28], "game": 4, "gamge": 28, "gamma": [0, 2, 8, 9, 10, 11, 13, 28, 30], "gamma1": 8, "gamma2": 8, "gamma_": [0, 28], "gamma_0": 10, "gamma_1": 10, "gamma_1x": 10, "gamma_i": [0, 8, 25, 28], "gamma_j": 13, "gamma_k": [13, 30], "gamma_m": 10, "gamma_x": [0, 28], "gap": [8, 31], "gate": [4, 12], "gather": [0, 1, 12, 29], "gaug": 12, "gaussbacksub": 22, "gaussian": [4, 5, 6, 8, 14, 18, 25, 28, 32], "gaussian_point": 14, "gaussian_rbf": 8, "gave": [13, 31], "gavra": 28, "gbc": 28, "gca": [2, 6, 8, 13], "gd": [1, 30], "gd_clf": 10, "gdclassiffiercgain": 10, "gdclassiffierconfus": 10, "gdclassiffierroc": 10, "gdm": 13, "gdregress": 10, "ge": [1, 5, 7, 25, 29, 30], "gen_loss": 4, "gen_tap": 4, "gender": [0, 28], "genener": 4, "gener": [0, 1, 2, 3, 5, 6, 8, 10, 11, 12, 13, 14, 15, 16, 18, 20, 22, 23, 25, 27, 29, 30, 31, 32], "generaliz": 16, "generallay": 12, "generate_and_save_imag": 4, "generate_imag": 4, "generate_latent_point": 4, "generate_simple_clustering_dataset": 14, "generated_imag": 4, "generator_loss": 4, "generator_loss_list": 4, "generator_model": 4, "generator_optim": 4, "genom": 21, "geodes": 11, "geoff": 31, "geometr": [0, 13, 28, 31], "geometri": 5, "georg": 27, "geotif": 6, "geq": [2, 5, 8, 9, 13, 29, 30, 31], "gerard": [], "geron": [0, 27, 28], "get": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 13, 15, 19, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "get_dummi": 9, "get_paramet": 2, "get_split": 9, "get_yaxi": 8, "get_yticklabel": 6, "getmask": [], "gh": 15, "giant": 31, "gibb": [21, 28], "gif": 4, "gini": 10, "gini_index": 9, "ginvers": 13, "git": [0, 15, 21, 28], "gitcdn": [], "giter": [13, 31], "github": [0, 20, 21, 23, 24, 26, 27, 28, 29], "gitignor": 15, "gitlab": [0, 15, 21, 23, 28], "give": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 14, 18, 19, 21, 23, 25, 28, 29, 30, 31, 32], "given": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 19, 22, 25, 28, 29, 30, 31, 32], "global": [6, 7, 13, 30, 31], "glorot": 1, "gmail": [], "gnew": 13, "go": [0, 1, 3, 5, 6, 8, 9, 11, 12, 13, 15, 16, 18, 28, 29, 30, 32], "goal": [0, 7, 9, 28], "goe": [0, 1, 2, 5, 6, 13, 14, 15, 19, 22, 28, 29, 30, 31, 32], "goessner": [], "golden": 13, "gone": [5, 29, 30], "gong": 1, "good": [1, 3, 4, 5, 6, 9, 10, 11, 13, 15, 18, 21, 25, 27, 29, 30, 31], "goodfellow": [4, 27, 28, 29, 30], "googl": [1, 4, 21, 28], "got": [1, 6, 23], "gotten": 28, "gov": 6, "govern": 28, "gp": 27, "gpu": [1, 13, 21, 28, 31], "grad": [2, 13, 31], "grad_analyt": 13, "grad_ol": 18, "grad_ridg": 18, "grade": [23, 24], "gradient": [0, 3, 4, 7, 8, 9, 12, 21, 28, 29], "gradient_desc": 31, "gradientboostingclassifi": 10, "gradientboostingregressor": 10, "gradients_of_discrimin": 4, "gradients_of_gener": 4, "gradienttap": 4, "gradual": [1, 14], "grai": [4, 6], "granger": [], "grant": [], "graph": [1, 9, 11, 12, 13, 16, 20, 30, 31], "graph_from_dot_data": 9, "graphic": [0, 1, 9, 15, 28], "grasp": 0, "gray_r": [1, 3], "grayscal": 3, "great": [5, 13, 15, 30, 31], "greater": [1, 7, 25, 29], "greatli": 13, "greedi": 9, "green": [0, 3, 9, 25], "grei": 4, "grid": [1, 3, 6, 7, 8, 12, 25, 29, 31, 32], "grossli": [13, 30], "ground": [0, 28], "group": [0, 6, 7, 9, 14, 15, 20, 21, 23, 24, 26, 28, 32], "groupbi": [0, 28], "grow": [1, 3, 9, 10, 31], "growth": [0, 28], "gru": 4, "guarante": [0, 4, 13, 25, 28, 29, 30, 31], "guess": [1, 4, 10, 13, 14, 30, 31], "guestrin": 10, "gui": 15, "guid": 1, "guidelin": 20, "g\u00f6ssner": [], "h": [0, 1, 5, 6, 8, 13, 15, 19, 25, 26, 27, 28, 29, 30, 31], "h1": 2, "h_": [0, 13, 28, 30, 31], "h_0": 31, "h_1": [2, 13, 30], "h_2": [2, 13, 30], "h_m": 10, "h_t": 31, "ha": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "haanen": [26, 28], "habit": [0, 29], "had": [0, 1, 6, 7, 13, 28, 30, 31, 32], "hadamard": [1, 12, 13, 31], "half": [1, 8, 9], "halv": 10, "hand": [0, 1, 2, 3, 5, 11, 12, 13, 21, 22, 23, 25, 26, 27, 28, 29, 30, 31], "handi": [3, 23], "handl": [0, 1, 2, 5, 9, 11, 15, 18, 21, 29, 30, 31], "handle_unknown": 9, "handsid": 12, "handwrit": 12, "handwritten": [1, 5], "happen": [1, 2, 3, 4, 5, 6, 10, 13, 25, 29, 30, 31], "hard": [1, 7, 8, 10, 13, 30, 31], "hardcopi": [21, 28], "harder": [0, 1, 19, 29], "harmon": 3, "hash": 31, "hasn": [], "hassl": [0, 21, 28], "hast": [21, 28], "hasti": [0, 6, 16, 17, 19, 20, 23, 27, 28, 29, 32], "hat": [0, 1, 5, 6, 7, 9, 10, 11, 12, 13, 16, 17, 18, 19, 22, 29, 30, 31, 32], "hauser": [], "have": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "have_sys_un_h": [], "haven": 1, "he": 7, "head": [4, 10, 25], "header": [0, 28], "heads_proba": 10, "health": [0, 29], "hear": [0, 13, 28, 31], "heart": [0, 7, 28], "heatmap": [0, 1, 3, 7, 17, 20, 28], "heavi": 31, "heavili": 0, "heavisid": 1, "height": [1, 3, 6, 29], "held": [13, 31], "help": [0, 1, 4, 12, 13, 15, 16, 23, 28, 31, 32], "helper": [4, 14], "henc": [0, 5, 6, 8, 9, 10, 12, 13, 28, 29, 30, 31, 32], "henrik": [26, 28], "her": 7, "here": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "hereaft": [0, 8, 12, 28], "herebi": [], "hermitian": 22, "hessenberg": 22, "hessian": [0, 2, 5, 13], "heterogen": [9, 10], "hex": [], "hi": 7, "hidden": [1, 3, 4, 12], "hidden_bia": 1, "hidden_bias_gradi": 1, "hidden_layer_s": [0, 1, 28], "hidden_neuron": 4, "hidden_weight": 1, "hidden_weights_gradi": 1, "hierarch": [5, 29, 30], "high": [0, 1, 2, 3, 4, 5, 6, 9, 10, 11, 13, 14, 21, 22, 23, 28, 29, 30, 31, 32], "higher": [0, 1, 3, 5, 6, 8, 13, 18, 23, 28, 29, 30, 31, 32], "highest": [1, 2], "highli": [0, 3, 4, 10, 19, 21, 22, 27, 28, 29, 30, 31], "highlight": [], "highwai": [], "hing": 8, "hint": [13, 15, 16, 29, 30], "hinton": 31, "hip": 21, "hire": 0, "hist": [4, 6, 7, 25, 32], "histogram": [6, 7, 25], "histor": [7, 11], "histori": [3, 4, 12, 15, 31], "hitherto": 5, "hjorth": [26, 28, 29, 30, 31, 32], "hobbi": 25, "hoc": [5, 29, 30], "hoff": 27, "hold": [1, 3, 6, 13, 14, 30, 31, 32], "holder": [0, 28], "holdgraf_evidence_2014": [], "home": [], "homepag": [23, 28], "homework": [6, 13, 30, 31], "homogen": [1, 3, 9, 10, 13, 31], "honchar": 2, "hopefulli": [0, 11, 15, 19, 25, 28, 31], "horizont": 11, "horlyk": [26, 28], "hors": [3, 7, 28], "hot": [1, 9], "hour": [1, 21, 24, 25, 26, 28, 31, 32], "how": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "howev": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 25, 28, 29, 30, 31, 32], "href": [], "hspace": [0, 4, 8, 10, 25, 28], "hstack": 1, "htf": 28, "html": [0, 16, 20, 21, 23, 24, 26, 27, 28, 29, 30, 31], "http": [0, 3, 4, 6, 13, 15, 16, 19, 20, 21, 22, 23, 24, 26, 27, 28, 29, 30, 31, 32], "huang": [0, 28], "huber": [0, 28], "huge": [1, 3, 4, 21, 31], "human": [0, 1, 3, 6, 9, 12, 29], "humid": 9, "hundr": 1, "hungri": 1, "hybrid": 24, "hydrogen": [0, 28], "hyperbol": [1, 4, 12], "hyperparam": 8, "hyperparamet": [3, 4, 5, 6, 9, 13, 18, 23, 29, 30, 31], "hyperplan": 11, "h\u00f8rlyk": [26, 28], "i": [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 29, 30, 31, 32], "i0": [0, 28], "i1": [0, 6, 8, 12, 28, 29, 31], "i2": [0, 8, 12, 28], "i3": [0, 12, 28], "i5": [0, 28], "i_": [13, 30, 31], "i_1": [5, 6, 32], "i_2": [5, 6, 32], "i_t": 31, "ian": 27, "ic": [1, 23], "id": [7, 13, 30, 31], "ida": [26, 28], "idea": [0, 1, 2, 3, 4, 6, 9, 10, 12, 13, 20, 22, 23, 29, 30, 31, 32], "ideal": [0, 2, 6, 8, 13, 25, 28, 31, 32], "idem": [6, 32], "ident": [5, 6, 12, 13, 17, 18, 22, 29, 30], "identical": 32, "identifi": [0, 1, 7, 9, 11, 12, 13, 14, 28, 29], "idx": [], "ieor": 25, "ifi": 27, "ifs": [21, 28], "ignor": [0, 1, 3, 9, 15, 29, 31], "ii": [22, 25], "iii": [22, 28], "ij": [0, 1, 3, 6, 8, 12, 14, 16, 22, 25, 28, 29, 31], "ik": [0, 22, 28, 29], "iki": [], "ill": 31, "illinoi": [], "illustr": [5, 7, 10, 12, 13, 14, 20, 21, 28], "ilsvrc": 31, "im": 6, "imag": [1, 3, 4, 6, 9, 11, 12, 14, 27, 28], "image_at_epoch_": 4, "image_batch": 4, "image_height": 3, "image_path": [0, 6, 7, 9, 28, 32], "image_width": 3, "imageio": 6, "imagenet": 31, "images_from_seed_imag": 4, "imagin": 1, "immedi": [0, 3, 4, 6, 21, 28, 31], "implement": [0, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 19, 20, 23, 25, 28, 29, 30, 31], "impli": [3, 5, 6, 7, 13, 22, 29, 30, 31, 32], "implicit": [3, 31], "implicitli": [11, 25], "import": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 23, 25, 31, 32], "importantli": 3, "importerror": [], "impos": [0, 6, 11, 12, 28], "imposs": [0, 5, 28, 29, 30], "impract": 31, "impress": [0, 12, 28], "improv": [0, 4, 5, 9, 10, 11, 13, 15, 23, 29, 30], "impur": 9, "imread": 6, "imshow": [1, 3, 4, 6], "in3050": [27, 28], "in3310": 28, "in4080": [27, 28], "in4300": [27, 28], "in4310": 27, "in5400": 3, "in5550": 27, "in_out_neuron": 4, "inaccur": [13, 30], "inact": 12, "inadequ": [0, 28], "inappropri": 31, "inch": [6, 29], "incident": [], "includ": [0, 1, 2, 3, 4, 5, 6, 7, 11, 12, 15, 16, 17, 18, 19, 20, 21, 25, 26, 27, 28, 29, 30, 32], "include_bia": [6, 9, 32], "incom": [12, 16], "incorrect": 1, "incoveni": 8, "increas": [0, 1, 3, 4, 5, 6, 9, 12, 13, 19, 23, 25, 28, 29, 31, 32], "increasingli": 25, "increment": 31, "ind": 6, "inde": [0, 2, 4, 5, 6, 13, 28, 29, 30], "indefinit": 4, "independ": [0, 5, 6, 7, 8, 12, 13, 25, 28, 29, 30, 31], "index": [0, 1, 3, 4, 10, 14, 21, 22, 23, 25, 27, 28], "index_col": [0, 28], "indic": [0, 1, 3, 4, 5, 6, 9, 10, 11, 13, 16, 23, 28, 29], "indirect": [], "indispens": [6, 32], "individu": [1, 6, 7, 10, 12, 25, 28, 29, 31, 32], "indu": [], "indx": 22, "indx1": 2, "indx2": 2, "indx3": 2, "ineffici": [3, 13], "inequ": [8, 13], "inequaltii": 30, "inertia": 13, "inexperi": [], "inf": [], "inf1000": [21, 28], "inf1100": [21, 28], "inf1100l": [21, 28], "inf1110": [21, 28], "inf3000": 28, "infeas": [9, 31], "infer": [0, 1, 4, 6, 27, 28, 32], "inferenc": 1, "infil": [0, 6, 7, 9, 28, 32], "infin": [5, 6, 7, 11, 19, 29, 30, 32], "infinit": [3, 31], "infinitesim": 25, "influenc": [6, 10, 18, 32], "influenti": 1, "info": 28, "inform": [0, 1, 3, 4, 6, 9, 11, 12, 13, 14, 22, 23, 27, 28, 30, 31, 32], "inforom": 15, "infrequ": 31, "infti": [3, 6, 13, 25, 30, 32], "ingeni": [13, 30, 31], "ingredi": [0, 9, 28], "inher": [6, 31, 32], "inherit": [22, 28, 31], "init": [], "initi": [0, 1, 2, 6, 10, 13, 14, 18, 22, 25, 28, 30, 31, 32], "inject": 14, "inlin": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 25, 28, 29, 30, 31, 32], "inner": [0, 13, 29], "innerhtml": [], "inp": 4, "inplac": 13, "input": [0, 1, 3, 4, 5, 6, 7, 8, 12, 13, 14, 16, 23, 25, 28, 29, 30, 31, 32], "input_dim": 1, "input_shap": [3, 4], "inputs": 1, "inputs_shuffl": [0, 1, 29], "inquiri": 20, "insert": [3, 5, 6, 8, 10, 25, 29, 30, 32], "insid": [4, 7], "insight": [0, 1, 5, 21, 28, 29, 30, 32], "insist": [6, 13, 29, 31], "inspir": [0, 1, 12, 23, 28], "instabl": 2, "instal": [0, 1, 5, 6, 9, 15, 20], "instanc": [0, 1, 2, 4, 6, 9, 11, 13, 16, 28, 29, 30, 31, 32], "instanti": 10, "instead": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 13, 14, 17, 20, 22, 25, 28, 29, 31, 32], "institut": 1, "instruct": [0, 1, 15], "int": [0, 1, 2, 3, 4, 5, 6, 11, 13, 14, 22, 25, 29, 31, 32], "int32": 10, "int_": [3, 6, 25, 32], "int_0": 25, "int_a": 25, "intak": [0, 29], "integ": [1, 2, 13, 14, 22, 25, 28], "integer_vector": 1, "integr": [3, 6, 25, 28, 32], "intellig": [0, 14, 27, 28], "intend": 10, "intens": [1, 18], "intention": 14, "interact": [0, 6, 9, 12, 21, 23, 28], "intercept": [0, 6, 8, 11, 13, 16, 17, 18, 19, 28, 29, 30, 31, 32], "intercept_": [0, 6, 8, 9, 13, 28, 29, 31], "interchang": [5, 12, 22], "interconnect": 1, "interesit": [], "interest": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 12, 19, 21, 23, 25, 28, 29, 30, 32], "interfac": [0, 1, 15, 22, 29], "interior": [0, 9, 28], "intermedi": [22, 29, 31], "intern": [1, 10, 12], "internation": [], "interpol": [1, 3, 4, 6, 12], "interpr": [5, 29, 30], "interpret": [0, 1, 6, 9, 10, 12, 13, 15, 16, 22, 23, 25], "interrupt": [], "interv": [0, 3, 5, 6, 7, 13, 19, 25, 28, 29, 30], "intial": [13, 30], "intract": [0, 4, 29], "intrins": [3, 11, 22, 25, 28], "intro": [21, 27, 28], "introduc": [0, 1, 5, 6, 8, 10, 12, 22, 23, 25, 28, 30, 31, 32], "introduct": [1, 2, 4, 13, 27, 29, 30, 31], "introductori": [0, 4, 22, 27, 28, 29], "intuit": [0, 5, 6, 8, 12, 13, 23, 28, 31, 32], "inv": [0, 5, 13, 17, 28, 29, 30, 31], "invalid": [], "invalu": [0, 13, 21, 28, 30], "invari": 1, "invd": 5, "inver": 8, "invers": [0, 3, 6, 13, 28, 29, 30, 31], "inverse_transform": 8, "invert": [0, 5, 7, 10, 13, 16, 18, 28, 31], "investig": [], "invh": [13, 31], "invok": 8, "involv": [0, 2, 6, 7, 11, 12, 28, 29, 31, 32], "io": [0, 21, 23, 24, 26, 27, 28, 29], "ion": [], "ip": [0, 8, 25, 28], "ipca": 11, "ipynb": [21, 28], "ipython": [0, 5, 7, 9, 11, 14, 21, 23, 28, 29], "iq": [6, 32], "iri": [8, 9], "irreduc": [6, 32], "irrelev": [5, 29, 30], "irrespect": [0, 28], "irvin": 23, "isaac": [], "isaacmus": [], "iseffici": [], "isn": 5, "isnul": [], "isomap": 11, "issu": [1, 9, 15, 22, 31], "it_arrai": 13, "item": [0, 13, 28], "items": [22, 28], "iter": [1, 2, 4, 6, 8, 13, 14, 18, 23, 25, 30, 31, 32], "its": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 20, 21, 22, 23, 25, 28, 30, 31, 32], "itself": [5, 6, 12, 23, 25, 28, 29, 32], "j": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 13, 14, 15, 16, 22, 23, 25, 27, 28, 29, 30, 31, 32], "j1": 22, "j_": 6, "j_41hld6ttu": 32, "j_lasso_sk": 6, "j_ridge_sk": 6, "j_sk": 6, "jackknif": [6, 21, 28, 32], "jacobian": [2, 13, 30], "janko": [], "jason": 4, "javascript": [], "jax": [21, 28, 31], "jeff": [], "jensen": [26, 28, 29, 30, 31, 32], "jerom": [19, 23, 27], "jhauser": [], "ji": [12, 22], "jit": 13, "jj": [0, 5, 6, 28, 32], "jk": [0, 1, 6, 12, 22, 28], "jl": [0, 28], "jm": 22, "jnp": 13, "job": [2, 8, 10, 15], "join": [0, 4, 6, 7, 9, 28, 32], "joint": [4, 5], "jonathan": [], "json": [], "judg": [13, 30], "judgement": 6, "julia": [21, 22, 23], "jump": [25, 31], "junk": 4, "jupit": 28, "jupyt": [0, 15, 16, 19, 21, 23, 27, 28, 32], "jupyterbook": [], "jupytext": [], "just": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 20, 21, 25, 28, 29, 30, 31, 32], "justif": 0, "justifi": [3, 10], "k": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 25, 26, 28, 29, 30, 31], "k0": 7, "k1": 7, "kaggl": [6, 23], "kappa_d": 25, "karl": [26, 28], "karush": 8, "katex": [], "katrin": [26, 28], "keep": [0, 1, 4, 5, 6, 11, 13, 14, 15, 18, 22, 23, 28, 29, 30, 31, 32], "keepdim": [1, 6, 10, 22, 32], "kei": [1, 3, 6, 12, 31], "kellei": [], "kenneth": [], "kept": [4, 6, 14, 32], "kera": [0, 4, 21, 23, 28], "kernel": [0, 1, 3, 21, 28, 29], "kernel_regular": [1, 3], "kernel_s": 4, "kernelpca": 11, "kev": [0, 28], "kevin": [27, 28], "kevinsheppard": [], "keyword": [18, 22, 28], "kfold": [6, 32], "kg": 1, "ki": 22, "kick": [1, 13, 31], "kiener": 2, "kilomet": [6, 29], "kim": [], "kind": [0, 2, 3, 4, 8, 12, 13, 14, 28, 29], "kingma": 31, "kj": [6, 12, 22, 29, 31], "kjm": [21, 28], "kkt": 8, "kl": 25, "km": [12, 28], "kmean": 14, "kmeanspoint": 14, "kn_k": 14, "know": [0, 1, 2, 5, 6, 8, 13, 15, 16, 17, 19, 20, 21, 28, 29, 30], "knowledg": [0, 21, 28], "known": [1, 3, 4, 5, 6, 7, 8, 9, 12, 18, 22, 23, 25, 27, 29, 31, 32], "kondev": [0, 28], "kp": 25, "kpca": 11, "kramdown": [], "kroneck": 14, "kt": [], "kuhn": 8, "kvalsund": [26, 28], "kwown": [0, 28], "l": [0, 1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 13, 22, 23, 25, 28, 30, 31], "l0": 7, "l1": [0, 1, 3, 7, 28], "l1_l2": [1, 3], "l1regl": 5, "l2": [1, 3], "l_": [22, 31], "l_1": 7, "l_2": [7, 13, 30, 31], "l_i": 31, "l_j": 12, "la": 13, "la_": [], "la_i": 12, "la_k": 12, "lab": [20, 21, 23, 28], "label": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 15, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "labelencod": [7, 10], "labels": [6, 8, 9], "labels_shuffl": [0, 1, 29], "laboratori": 24, "lack": [0, 28, 31], "lagari": 2, "lagrang": [8, 11], "lam": 18, "lambda": [0, 1, 2, 3, 5, 6, 7, 8, 10, 12, 13, 17, 18, 19, 20, 23, 25, 28, 29, 30, 31, 32], "lambda_": 11, "lambda_0": 11, "lambda_1": [5, 8, 11, 29, 30], "lambda_2": [8, 11], "lambda_i": [8, 11], "lambda_iy_i": 8, "lambda_jy_iy_j": 8, "lambda_k": 8, "lambda_n": [5, 8, 29, 30], "lamda": 1, "land": 8, "landmark": 8, "landscap": [13, 18, 30, 31], "langl": [0, 6, 11, 25, 28, 29], "languag": [0, 1, 4, 8, 21, 22, 23, 27, 28], "lapack": [22, 28], "laplac": 5, "laptop": [15, 21], "larg": [0, 1, 2, 4, 5, 6, 8, 9, 10, 11, 13, 18, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "larger": [0, 3, 5, 6, 8, 10, 11, 13, 17, 25, 28, 29, 30, 31, 32], "largest": [4, 8, 11], "lasso": [0, 7, 21, 28, 31, 32], "lasso_sk": 6, "last": [0, 1, 3, 4, 5, 6, 7, 8, 12, 16, 17, 19, 22, 23, 25, 26, 28, 30, 32], "latent": 4, "latent_dim": 4, "latent_point": 4, "latent_space_value_rang": 4, "later": [0, 1, 4, 7, 8, 12, 13, 14, 15, 19, 21, 23, 28, 31], "latest": [4, 15, 21], "latest_checkpoint": 4, "latex": [20, 28], "latexcodec": [], "latter": [0, 3, 6, 7, 8, 11, 13, 22, 25, 28, 29, 30, 31, 32], "lattic": 12, "law": 0, "layer": [0, 4, 13, 28, 31], "lbfg": [7, 9, 10], "lc_messag": [], "lcc": [5, 6, 32], "lda": 11, "ldot": [0, 6, 11, 23, 28, 32], "le": [5, 7, 10, 13, 17, 25, 29, 30, 31], "lead": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 22, 25, 28, 29, 30, 31, 32], "leaf": 9, "leaki": 1, "leakyrelu": 4, "lear": [13, 30], "learn": [3, 4, 5, 6, 7, 8, 9, 10, 12, 22, 26, 27], "learnabl": 3, "learner": 10, "learnig": 28, "learning_r": [8, 10], "learning_rate_init": [0, 1, 28], "learning_schedul": [13, 31], "learnt": 23, "least": [0, 7, 8, 10, 11, 17, 18, 21, 22, 25, 32], "leat": [13, 31], "leav": [0, 1, 3, 5, 6, 9, 11, 28, 30, 32], "lectur": [0, 1, 5, 10, 11, 12, 13, 21, 22, 23, 24, 26, 27, 29], "lecturenot": [0, 21, 23, 27, 28], "left": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "leftarrow": [8, 12], "legend": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 15, 28, 29, 30, 31, 32], "leinonen": 28, "len": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 16, 17, 22, 28, 29, 30, 31, 32], "length": [0, 1, 3, 4, 8, 9, 13, 16, 21, 28, 29, 30, 31], "length_of_sequ": 4, "leq": [0, 5, 7, 8, 13, 14, 25, 28, 29, 30, 31], "less": [0, 1, 3, 4, 5, 6, 8, 9, 13, 21, 25, 28, 29, 30, 31, 32], "lessen": 1, "let": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 19, 22, 25, 28, 29, 30, 31, 32], "letter": [0, 16, 22, 25, 28, 29], "level": [0, 1, 5, 6, 9, 21, 22, 23, 24, 26, 28, 31, 32], "leverag": 31, "lexer": [], "li": [8, 11], "liabil": [], "liabl": [], "lib": [], "liberti": 31, "liblinear": 10, "librari": [0, 1, 2, 3, 4, 5, 6, 9, 10, 11, 22, 23, 25, 27, 29, 30, 31], "licenc": [], "licens": [0, 1, 21, 23, 28], "lie": [0, 6, 11, 25, 28, 29, 32], "life": [0, 1, 8, 12, 28], "lifetim": 13, "light": [], "like": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 15, 16, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "likelihood": [0, 1, 5, 9, 28, 29], "lim_": 25, "limit": [0, 5, 6, 8, 12, 22, 23, 28, 29], "lin_clf": 8, "lin_model": [], "lin_reg": 9, "linalg": [0, 2, 5, 6, 8, 11, 13, 17, 22, 25, 28, 29, 30, 31], "line": [0, 3, 6, 8, 11, 13, 15, 16, 20, 28, 30, 31, 32], "line1": 8, "line2": 8, "line2d": [], "line3": 8, "line_model": 15, "line_ms": 15, "line_predict": 15, "linear": [1, 3, 5, 6, 7, 9, 10, 11, 12, 16, 17, 18, 19, 21, 23, 25, 31, 32], "linear_model": [0, 5, 6, 7, 8, 9, 10, 11, 13, 15, 16, 19, 28, 29, 30, 31, 32], "linear_regress": [6, 32], "linearli": [5, 29, 30, 31], "linearloc": [6, 13, 30, 31], "linearregress": [0, 6, 7, 9, 15, 16, 19, 28, 29, 31, 32], "linearsvc": 8, "lineat": 30, "liner": [1, 3], "linerar": 10, "linewidth": [0, 2, 4, 6, 8, 9, 10, 32], "link": [0, 4, 9, 12, 15, 20, 21, 23, 24, 26, 28], "linlag": 5, "linpack": [22, 28], "linreg": [0, 28], "linspac": [0, 2, 3, 4, 6, 8, 9, 10, 13, 16, 17, 19, 22, 25, 28, 29, 31, 32], "linu": 4, "linux": [0, 1, 21, 23, 28], "liquid": [0, 28], "list": [1, 2, 3, 4, 9, 15, 21, 23, 28, 31], "listedcolormap": [9, 10], "literatur": [1, 7, 14, 27, 32], "littl": [1, 3, 9, 12, 31], "live": [8, 16], "ll": [0, 18, 25, 28, 29], "lle": [0, 29], "llm": 20, "lloyd": [4, 14], "lmb": [0, 2, 5, 6, 29, 30, 31, 32], "lmbd": [0, 1, 3, 28], "lmbd_val": [0, 1, 3, 28], "lmbda": [13, 30, 31], "ln": [1, 13, 30], "load": [1, 4, 6, 7, 9, 10, 31], "load_boston": [], "load_breast_canc": [1, 7, 9, 10, 11], "load_data": [3, 4], "load_digit": [1, 3], "load_iri": [8, 9], "loc": [3, 6, 7, 8, 9, 10, 28, 32], "local": [0, 1, 3, 7, 12, 13, 15, 29, 30, 31], "locat": [2, 3, 8, 15], "log": [0, 1, 2, 4, 5, 6, 7, 9, 10, 11, 13, 15, 20, 22, 23, 28, 31, 32], "log10": [0, 5, 6, 29, 30, 31, 32], "log_": [0, 28], "log_clf": 10, "logarithm": [0, 5, 7, 17, 22, 28, 32], "logbook": 23, "logic": [0, 1, 9, 28], "logical_or": [], "login": 15, "logist": [0, 1, 2, 8, 9, 10, 11, 12, 13, 21, 29, 30, 31], "logisticregress": [7, 9, 10, 11], "logit": 7, "logreg": [7, 9, 10, 11], "logspac": [0, 1, 3, 5, 6, 28, 29, 30, 31, 32], "long": [0, 1, 3, 4, 12, 13, 28, 30, 31], "longer": [2, 3, 8, 10, 14, 22, 25, 28, 31], "loocv": [6, 32], "look": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 16, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "loop": [1, 4, 6, 10, 12, 14, 16, 17, 18, 21, 22, 28, 31, 32], "lose": 1, "loss": [0, 1, 3, 4, 5, 6, 7, 8, 10, 11, 13, 18, 22, 23, 28, 32], "loss_fil": 4, "lossfil": 4, "lost": 4, "lot": [1, 4, 6, 16, 19, 20, 31, 32], "low": [0, 6, 9, 10, 11, 23, 28, 29, 32], "lower": [0, 1, 3, 6, 9, 10, 16, 22, 29, 31], "lowercas": [22, 28], "lowest": [9, 13, 25, 31], "lr": [1, 3, 4, 10], "lstat": [], "lstm": 4, "lstm_2layer": 4, "lstsq": [0, 28, 29], "lt": [6, 32], "lu": [0, 5, 28, 29, 30], "lubksb": 22, "luckili": 2, "ludcmp": 22, "lux": 22, "lvert": 1, "lw": [0, 28], "m": [0, 1, 2, 3, 5, 6, 8, 9, 10, 11, 12, 13, 15, 22, 25, 26, 27, 28, 29, 30, 31, 32], "m_": [9, 12], "m_0": 31, "m_1": 14, "m_h": [0, 28], "m_k": 14, "m_l": 12, "m_n": [0, 28], "m_p": [0, 28], "m_t": [13, 31], "ma": 11, "machin": [1, 3, 4, 5, 6, 7, 9, 10, 11, 12, 15, 16, 22, 27, 29, 31, 32], "machinelearn": [0, 6, 16, 20, 21, 23, 24, 26, 27, 28, 29, 30], "machineri": [], "mackai": 27, "macro": [], "made": [0, 1, 3, 4, 5, 6, 7, 9, 11, 12, 23, 28, 29, 31], "mae": [0, 28], "magic": 4, "magnitud": [1, 6, 7, 13, 29, 31], "mai": [0, 1, 2, 3, 5, 6, 7, 8, 9, 11, 12, 13, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "mail": [24, 26], "main": [0, 1, 3, 4, 5, 6, 7, 9, 22, 23, 27, 29, 30, 31], "mainli": [0, 5, 6, 7, 9, 28, 29, 32], "maintain": [6, 31, 32], "major": [1, 6, 9, 10, 13, 22, 28, 30, 31, 32], "make": [1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 15, 16, 18, 19, 21, 22, 23, 25, 27, 28, 30, 31, 32], "make_axes_locat": 6, "make_moon": [8, 9, 10], "make_pipelin": [0, 6, 10, 29, 32], "makedir": [0, 6, 7, 9, 28, 32], "malcondit": 22, "malign": [1, 7, 9], "mammographi": 5, "manag": [0, 2, 3, 15, 21, 23, 28, 31], "mandatori": [26, 28], "mani": [0, 1, 3, 4, 5, 6, 7, 8, 9, 11, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "manifold": 11, "manner": 3, "manual": [6, 29, 31], "map": [0, 1, 2, 6, 7, 8, 11, 12, 14, 25, 28], "marc": 29, "marchant": [], "margin": [0, 5, 8], "marit": [0, 28], "mark": 28, "markdownfil": [], "markdownit": [], "markdownitdeflist": [], "markedli": [], "marker": [7, 22, 28], "markov": [21, 28], "markup": [], "marsaglia": 25, "mask_or": [], "masked_arrai": [], "maskedrecord": [], "mass": [0, 1, 5, 13, 29, 30], "massag": [0, 28], "masses2016": [0, 28], "masses2016ol": [0, 28], "masses2016tre": 0, "masseval2016": [0, 28], "master": [24, 26], "mat": [21, 28], "mat1100": [21, 28], "mat1110": [21, 28], "mat1120": [21, 28], "match": [1, 4, 5, 13, 14, 15, 29, 30, 31], "materi": [4, 5, 7, 13, 15, 22, 24, 26], "math": [3, 7, 12, 13, 22, 25, 27, 28, 31], "mathbb": [0, 4, 5, 6, 7, 8, 11, 12, 13, 14, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "mathbf": [0, 5, 6, 7, 8, 13, 19, 22, 23, 28, 29, 30, 31, 32], "mathcal": [1, 5, 6, 7, 13, 23, 32], "matheemat": 3, "mathemat": [0, 6, 11, 12, 13, 21, 22, 25, 27, 28, 31], "mathemati": 28, "mathrm": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 18, 19, 23, 25, 28, 29, 30, 31, 32], "matmul": [1, 2, 5], "matnat": 27, "matplotlib": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "matplotlibrc": [], "matric": [0, 1, 3, 4, 6, 7, 8, 11, 13, 16, 17, 21, 29, 30], "matrix": [0, 2, 3, 4, 6, 7, 8, 10, 13, 17, 18, 19, 23, 25, 32], "matshow": 1, "matter": [2, 3, 13, 29, 30, 31], "matthia": [], "max": [0, 1, 2, 3, 4, 9, 10, 12, 13, 26, 28, 30, 31], "max_depth": [0, 9, 10], "max_diff": 2, "max_diff1": 2, "max_diff2": 2, "max_it": [0, 1, 8, 13, 28], "max_iter": 14, "max_leaf_nod": 10, "max_sampl": 10, "maxdegre": [0, 6, 10, 29, 32], "maxdepth": 10, "maxim": [1, 4, 5, 7, 8, 11, 32], "maximum": [0, 2, 3, 5, 7, 8, 9, 10, 13, 14, 28, 29, 30, 31], "maxpolydegre": [5, 6, 29, 30, 31, 32], "maxpooling2d": 3, "mbox": [5, 6, 29, 30, 32], "mcculloch": 12, "md": 11, "mdoel": 4, "me": [], "mean": [1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 21, 22, 23, 25, 28, 31, 32], "mean_absolute_error": [0, 28], "mean_divisor": 14, "mean_i": 25, "mean_matrix": 14, "mean_squared_error": [0, 4, 6, 7, 10, 15, 19, 28, 29, 32], "mean_squared_log_error": [0, 28], "mean_vector": 14, "mean_x": 25, "meaning": [0, 4, 7, 28], "meansquarederror": [0, 28], "meant": [3, 7, 10, 13], "meanwhil": 31, "measur": [0, 1, 2, 5, 6, 9, 11, 12, 14, 16, 18, 23, 25, 28, 29, 31, 32], "mechan": [0, 4, 25, 28, 31], "median": [0, 28, 29, 31], "medicin": 12, "medium": [4, 8, 13, 31], "medv": [], "meet": [0, 26], "mehta": [0, 28, 29, 30], "member": 20, "memori": [3, 4, 11, 12, 13, 18, 22], "mentat": [], "mention": [0, 12, 13, 23, 25, 28, 30, 31], "merchant": [], "mere": [0, 23], "merg": [], "meshgrid": [2, 5, 6, 8, 9, 10, 11], "mess": 15, "messag": [5, 13], "messi": 2, "met": [0, 3, 8, 29], "meta": [], "meteorolog": 9, "meter": [6, 29], "method": [0, 1, 2, 3, 4, 5, 7, 8, 11, 12, 14, 15, 16, 17, 18, 19, 20, 21, 22, 25, 27, 29], "metion": 6, "metric": [0, 1, 3, 6, 7, 9, 10, 14, 15, 28, 29, 32], "metropoli": [21, 28], "mev": [0, 25, 28], "mgd": [13, 31], "mglearn": [21, 28], "mgrid": 13, "mhjensen": [], "mi": 10, "mia": [26, 28], "michael": [], "microsoft": 27, "mid": 1, "midel": 4, "midnight": 15, "midpoint": 9, "might": [0, 1, 2, 4, 6, 9, 13, 15, 17, 18, 29, 30, 31], "migth": 17, "mild": 9, "millimet": [6, 29], "million": [0, 28, 29, 31], "mimic": 12, "min": [0, 2, 5, 8, 9, 30], "min_": [0, 2, 5, 14, 17, 28, 29, 30], "min_samples_leaf": 9, "mind": [0, 6, 13, 15, 18, 28, 29, 30, 31, 32], "mindboard": 4, "mine": [21, 28], "mini": [1, 11, 12, 13, 30], "minibatch": [1, 11, 13], "minibathc": [13, 31], "miniforge3": [], "minim": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 29, 30, 31, 32], "minima": [0, 1, 7, 13, 28, 30, 31], "minimum": [0, 1, 2, 6, 8, 9, 11, 13, 29, 30, 31, 32], "minmaxscal": [0, 29, 31], "minor": 25, "minst": 1, "minu": 7, "mirjalili": 28, "mirror": 9, "misc": 6, "misclassif": [8, 9, 10], "misclassifi": [8, 10], "miser": 0, "mismatch": 1, "miss": [7, 10], "mistak": [4, 19], "mit": 27, "mitig": 31, "mix": [1, 2, 28], "mixtur": [13, 31], "mk": [9, 22], "mkdir": [0, 6, 7, 9, 28, 32], "ml": [0, 1, 10, 13, 22, 23, 29, 30, 31], "mlab": 25, "mle": [5, 7], "mlp": 1, "mlpclassifi": 1, "mlpregressor": [0, 28], "mm": 22, "mml": 29, "mn": [12, 25], "mnist": [1, 11], "mo": [], "mod": 25, "mode": [24, 26, 28], "model": [2, 3, 5, 7, 8, 9, 10, 11, 13, 14, 16, 18, 19, 20, 21, 23, 25, 27, 29, 30, 31, 32], "model_select": [0, 1, 3, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "moder": [10, 31], "modern": [0, 6, 7, 21, 28, 31, 32], "modest": 31, "modif": [2, 12, 13], "modifi": [0, 1, 3, 5, 7, 8, 10, 12, 13, 28, 29, 30, 31], "modul": [0, 16, 22, 28], "modular": 25, "modulo": 25, "moe": [11, 29], "moment": [5, 6, 13, 25, 32], "mondai": [26, 28], "monitor": [13, 31], "monoton": [5, 12, 25, 32], "mont": [0, 6, 21, 25, 27, 28, 32], "montli": 16, "moor": [5, 6], "more": [0, 1, 2, 4, 5, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 21, 25], "moreov": [0, 3], "morten": [26, 28, 29, 30, 31, 32], "mortenhj": 28, "most": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 21, 23, 25, 28, 29, 30, 31, 32], "mostli": [1, 11, 18, 31], "motion": [0, 13], "motiv": [1, 4], "moulin": 31, "move": [0, 4, 5, 6, 7, 9, 12, 13, 14, 15, 16, 23, 25, 29, 30, 32], "mpl": [7, 28], "mpl_toolkit": [2, 6, 13, 30, 31], "mplot3d": [2, 6, 13, 30, 31], "mplregressor": 1, "mr_": [], "mrecord": [], "mse": [0, 4, 5, 6, 9, 10, 15, 16, 17, 19, 20, 23, 28, 29, 30, 31, 32], "mse_simpletre": 10, "mselassopredict": [5, 30], "mselassotrain": [5, 30], "mseownridgepredict": [6, 29, 30, 31], "msepredict": [5, 30], "mseridgepredict": [0, 5, 6, 29, 30, 31], "msetrain": [5, 30], "msg": [], "msle": [0, 28], "mt": [7, 12], "mu": [0, 6, 11, 13, 25, 28, 31, 32], "mu0": 25, "mu1": 25, "mu2": 25, "mu_": [6, 25, 29, 31, 32], "mu_i": [6, 29, 31], "mu_n": 11, "mu_x": 25, "much": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 15, 20, 22, 23, 25, 28, 29, 30, 31, 32], "multi": [0, 1, 3, 7, 21, 28], "multiclass": [1, 7], "multidimension": [11, 12, 28], "multilay": 1, "multinomi": 7, "multipl": [2, 4, 5, 6, 7, 12, 13, 15, 25, 29, 30, 31, 32], "multipli": [3, 5, 6, 11, 13, 18, 22, 25, 29, 30, 31], "multiplum": 8, "multivari": [0, 2, 10, 11, 21, 25, 28], "multivariate_norm": [11, 14], "multpli": 16, "murphi": [11, 27, 28], "muse": [], "must": [1, 2, 5, 6, 8, 10, 12, 13, 14, 15, 20, 25, 29, 30, 31, 32], "mutat": 7, "mutual": [1, 3, 6, 13, 32], "mx_": 25, "my": 28, "myenv": [], "myriad": [0, 21, 28], "myself": [], "mz1": 25, "mz2": 25, "m\u00f8svatn": 6, "n": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "n1": 22, "n2": 22, "n8grai": [], "n_": [1, 2, 3, 8, 12, 25], "n_0": [12, 25], "n_boostrap": [6, 10, 32], "n_bootstrap": [6, 32], "n_categori": [1, 3], "n_cluster": 14, "n_compon": 11, "n_epoch": [13, 31], "n_estim": 10, "n_examples_to_gener": 4, "n_featur": [1, 18], "n_filter": 3, "n_hidden": 2, "n_hidden_neuron": [0, 1, 28], "n_i": 25, "n_input": [0, 1, 3, 29], "n_instanc": 9, "n_iter": 31, "n_job": 10, "n_k": 14, "n_l": [12, 25], "n_layer": 1, "n_m": 9, "n_neuron": 1, "n_neurons_connect": 3, "n_neurons_layer1": 1, "n_neurons_layer2": 1, "n_point": 14, "n_sampl": [6, 8, 9, 10, 14, 18, 32], "n_split": [6, 32], "n_step": 4, "n_t": 2, "n_x": 2, "nabla": [1, 13, 30, 31], "nabla_": [2, 13, 30, 31], "nabla_w": 13, "nag": 13, "naimi": [0, 28], "naiv": 7, "naive_kmean": 14, "name": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 15, 18, 20, 21, 22, 23, 25, 26, 28, 29, 30, 32], "namespac": [], "nan": [], "narrow": [13, 31], "nathaniel": [], "nation": [1, 5], "nativ": [21, 28], "natur": [0, 1, 4, 8, 9, 12, 13, 23, 25, 27, 28, 30, 31], "navier": 12, "navig": [15, 31], "nb": 25, "nb_": 22, "nbconvert": 28, "nd": 14, "ndarrai": 6, "ne": [9, 10, 22, 25, 29, 30], "nearest": [1, 3, 6, 11], "nearli": [13, 30], "neat": 28, "neccesari": [6, 32], "necess": 2, "necessari": [0, 1, 3, 4, 8, 14, 18, 28], "necessarili": [0, 4, 11, 25, 28], "necesserali": 5, "neck": 7, "need": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 22, 25, 29, 30, 31, 32], "neg": [0, 1, 3, 5, 6, 7, 10, 13, 22, 25, 28, 30, 32], "neg_mean_squared_error": [6, 32], "neglect": [25, 31], "neglig": 25, "neighbor": [3, 6, 11], "neither": [4, 13, 31], "neq": [13, 14, 25, 30], "nervou": 12, "nest": [9, 12], "nesterov": 13, "net": [2, 4, 12], "netlib": [22, 28], "network": [0, 9, 13, 21, 27, 29], "neural": [0, 13, 21, 27, 29], "neural_network": [0, 1, 2, 28], "neuralnetwork": 1, "neuron": [1, 2, 3, 4, 12], "neutral": [0, 28], "neutron": [0, 28], "never": [1, 4, 6, 9, 25, 32], "new": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 17, 20, 22, 28, 29, 30, 31], "new_chang": [13, 31], "new_hobbit": 28, "new_ma": [], "newaxi": [0, 3, 6, 9, 32], "newli": [0, 28], "newlin": [], "newton": [1, 7, 8, 13, 25], "next": [0, 1, 2, 3, 4, 5, 6, 8, 9, 13, 14, 15, 16, 28, 29, 30, 31, 32], "next_guess": 13, "next_input": 4, "ng": 1, "ni": 14, "nice": [0, 1, 5, 11, 28, 29, 30], "nicer": [18, 31], "nip": 31, "niter": [13, 30, 31], "nitric": [], "nlambda": [0, 5, 6, 29, 30, 31, 32], "nlp": 27, "nm": 25, "nm_n": [0, 28], "nmse": [6, 32], "nn": [2, 5, 6, 12, 22, 28, 32], "nn_model": 1, "nnmin": 2, "node": [1, 3, 9, 10, 12], "nois": [0, 4, 5, 6, 8, 9, 10, 13, 18, 19, 23, 28, 29, 30, 31, 32], "noise_dimens": 4, "noisi": [1, 6, 23, 31, 32], "nomask": [], "non": [0, 1, 3, 5, 6, 7, 9, 10, 11, 12, 13, 14, 18, 22, 25, 28, 29, 30, 32], "nondifferenti": 31, "none": [0, 1, 2, 4, 5, 9, 10, 13, 25, 28, 29], "noninfring": [], "nonlinear": [3, 6, 8, 9, 11, 12, 32], "nonneg": [6, 9, 13, 30, 32], "nonparametr": 6, "nonsens": 25, "nonsingular": 22, "nonumb": [3, 7, 8, 13, 22], "nor": [1, 4, 13, 31], "norm": [0, 1, 5, 6, 8, 11, 13, 18, 28, 29, 30, 31, 32], "normal": [3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31], "normali": [22, 28], "norwai": [6, 23, 28, 30, 31, 32], "notabl": [], "notat": [0, 2, 5, 6, 13, 14, 25, 28, 29, 30, 32], "note": [0, 1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 14, 15, 16, 18, 21, 22, 25, 27, 28, 31, 32], "notebook": [0, 1, 3, 9, 15, 16, 19, 20, 21, 23, 28, 32], "noteworthi": 31, "noth": [1, 2, 5, 8, 12, 14, 25, 29, 30], "notic": [4, 5, 12, 13, 22, 25, 28], "notion": 3, "novel": [3, 6, 10, 28], "novemb": [1, 26, 28], "now": [0, 2, 4, 5, 6, 7, 8, 10, 11, 12, 14, 15, 16, 19, 21, 22, 23, 25, 28, 29], "nowadai": [0, 1, 3, 9, 21, 28], "nox": [], "np": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 25, 28, 29, 30, 31, 32], "npm": [], "npr": 2, "nsampl": [6, 32], "nt": 2, "nu": 25, "nuclear": [5, 29, 30], "nuclei": [0, 25, 28], "nucleon": [0, 28], "nucleu": [0, 28], "num": 4, "num_coordin": 2, "num_hidden_neuron": 2, "num_it": [2, 18], "num_neuron": 2, "num_neurons_hidden": 2, "num_point": 2, "num_tre": 10, "num_valu": 2, "number": [1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 18, 19, 22, 23, 24, 26, 28, 30, 32], "numberid": 7, "numberparamet": 3, "numer": [0, 5, 6, 9, 10, 11, 12, 13, 21, 22, 27, 28, 29, 30, 31, 32], "numpi": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 21, 23, 25, 29, 30, 31, 32], "numpydocstr": [], "nunmpi": [5, 29], "nve_frngahw": 30, "nx": 2, "ny": 25, "o": [0, 1, 4, 5, 6, 7, 8, 9, 11, 22, 26, 27, 28, 29, 30, 31, 32], "obei": [6, 11, 13, 29, 31], "object": [0, 1, 4, 8, 10, 15, 19, 22, 28, 31], "obliqu": [5, 29, 30], "observ": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 25, 28, 30, 31, 32], "obtain": [0, 1, 5, 6, 7, 8, 9, 10, 12, 13, 14, 17, 22, 23, 25, 28, 29, 30, 31, 32], "obviou": [5, 6, 11, 25, 29, 30], "obviouli": 28, "obvious": [0, 4, 5, 6, 22, 28, 32], "oc": [29, 30], "occupi": [], "occur": [0, 6, 8, 9, 22, 25, 28], "octob": [26, 28], "od": 0, "odd": [0, 3, 7, 28, 29, 31], "odenum": 2, "odesi": 2, "oen": 0, "off": [1, 3, 4, 5, 9, 13, 20, 25, 31, 32], "offer": [6, 11, 21, 22, 24, 26, 28, 32], "offic": [26, 28], "offici": [24, 28], "often": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "ofter": [22, 28], "ol": [0, 13, 17, 19, 29, 31], "old": [1, 5, 10, 13, 15, 18], "old_ma": [], "oliph": [], "ols_paramet": 16, "ols_sk": 6, "ols_svd": 6, "olsbeta": 30, "olstheta": [0, 5], "omega": [2, 3, 6], "omega_0": 3, "omit": [0, 5, 28, 29, 30, 32], "onc": [1, 6, 9, 11, 13, 20, 32], "one": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 19, 20, 21, 22, 23, 25, 26, 28, 29, 31, 32], "onehot": 1, "onehot_vector": 1, "onehotencod": 9, "ones": [0, 2, 5, 6, 8, 9, 10, 11, 13, 16, 18, 22, 23, 28, 29, 30, 31, 32], "ones_lik": 4, "ong": 29, "onl": 3, "onli": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "onlin": [11, 15, 20, 24, 31], "onto": [5, 11, 29, 30], "open": [0, 1, 4, 6, 7, 9, 15, 21, 23, 24, 26, 28, 32], "oper": [0, 1, 3, 5, 6, 10, 11, 12, 13, 15, 16, 21, 25, 28, 29, 30, 31, 32], "operation": 25, "oplu": 25, "opmiz": [13, 31], "opportun": 0, "oppos": [6, 13], "opposit": [1, 5, 8, 29, 30], "opt": [1, 5, 23, 28, 30], "optim": [0, 2, 3, 4, 5, 6, 7, 9, 10, 11, 14, 16, 17, 19, 23, 32], "optimis": [1, 3], "option": [0, 1, 3, 5, 6, 8, 11, 15, 18, 22, 29, 31, 32], "optmiz": [1, 8, 13, 29], "oral": 28, "orang": 0, "order": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 15, 22, 23, 25, 28, 29, 30, 32], "ordinari": [0, 2, 3, 7, 11, 13, 17, 18, 21, 32], "oreilli": [27, 28], "org": [0, 3, 4, 16, 20, 21, 22, 23, 27, 28, 29, 30, 31], "organ": [6, 7, 10, 22, 32], "orient": [1, 5, 25, 29, 30], "origin": [0, 3, 5, 6, 8, 11, 12, 13, 15, 22, 28, 29, 30, 31, 32], "orthogn": [5, 29, 30], "orthogon": [0, 5, 6, 8, 11, 13, 22, 28, 29, 30], "orthonorm": [5, 29, 30], "os": [26, 28], "oscar": 1, "oscil": [3, 13, 31], "oskar": 28, "oskarlei": 28, "osl": 18, "oslo": [0, 21, 23, 24, 26, 28, 29, 30, 31, 32], "osx": [0, 21, 23, 28], "other": [0, 1, 2, 3, 5, 6, 7, 8, 10, 13, 14, 16, 19, 21, 24, 25, 26, 27, 29, 30, 31, 32], "otherwis": [0, 1, 4, 7, 13, 22, 28, 31], "ouput": [5, 7, 12, 32], "our": [1, 2, 3, 6, 7, 8, 9, 10, 12, 14, 15, 16, 17, 18, 19, 21, 22, 25, 31, 32], "ourmodel": 0, "ourselv": [0, 5, 6, 8, 11, 13, 28, 29, 30, 32], "out": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 21, 22, 23, 25, 28, 29, 31, 32], "out_fil": 9, "outcom": [0, 7, 9, 10, 12, 25, 29], "outdoor": 9, "outer": [6, 12, 13], "outfil": 4, "outlier": [0, 8, 28, 29, 31], "outlin": [6, 10, 11, 32], "outlook": 9, "outperform": [10, 31], "output": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 22, 23, 25, 28, 29, 30, 31, 32], "output_bia": 1, "output_bias_gradi": 1, "output_shap": 4, "output_weight": 1, "output_weights_gradi": 1, "outputlayer1": 12, "outputlayer2": 12, "outsid": 4, "over": [0, 1, 3, 4, 5, 6, 9, 10, 12, 13, 15, 16, 19, 22, 23, 28, 29, 30, 31, 32], "over1": 13, "overal": [1, 10, 31], "overcast": 9, "overcom": [12, 13], "overdetermin": [0, 28], "overfit": [0, 1, 3, 6, 9, 10, 13, 31, 32], "overflow": [5, 31, 32], "overhead": 12, "overlap": [3, 7, 8, 9], "overleaf": 20, "overlin": [0, 5, 6, 9, 10, 11, 14, 22, 28, 29, 31], "overshoot": 31, "overst": 0, "overtrain": 4, "overview": [3, 20], "own": [4, 5, 6, 8, 12, 13, 16, 18, 21, 22, 30, 31, 32], "owner": [], "ownmsepredict": 0, "ownmsetrain": 0, "ownridgebeta": 29, "ownridgetheta": [0, 6, 29, 30, 31], "ownypredictridg": 0, "ownytilderidg": 0, "ox": [], "oxid": [], "p": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 25, 28, 29, 30, 31, 32], "p0": 2, "p1": 2, "p_": [2, 4, 8, 9], "p_hidden": 2, "p_i": [5, 25], "p_j": 25, "p_n": 25, "p_output": 2, "p_x": 25, "pack": [0, 28], "packag": [0, 1, 3, 4, 5, 8, 11, 13, 15, 20, 21, 23, 25, 29, 30, 31], "packtpub": 28, "packtpublish": 28, "pad": [3, 4], "page": [0, 21, 28, 30, 31, 32], "pai": [0, 1, 9, 13, 15, 31], "pair": [0, 2, 3, 9, 21, 25, 28], "paltform": 15, "panda": [0, 4, 5, 6, 7, 9, 11, 21, 23, 30, 31, 32], "pandoc": [], "panel": 28, "paper": [1, 31], "paper_fil": 31, "paradigm": [0, 28], "paragraph": 20, "parallel": [10, 13, 21, 22, 28], "param": 2, "paramat": 2, "paramet": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 16, 17, 18, 19, 23, 25, 30, 31, 32], "parameter": [0, 6, 10, 28, 29], "parametr": [0, 6, 28, 29, 32], "paramt": [3, 5, 32], "parser": [], "part": [0, 1, 3, 5, 6, 10, 17, 19, 20, 22, 24, 25, 26, 28, 29, 32], "partial": [0, 1, 5, 6, 7, 8, 10, 11, 12, 13, 16, 25, 28, 29, 30, 31], "particip": [15, 21, 24, 26, 28], "particl": [0, 4, 13, 25, 28], "particular": [0, 1, 2, 3, 5, 6, 9, 10, 11, 12, 13, 16, 23, 25, 27, 28, 29, 30, 31, 32], "particularli": [5, 6, 8, 11, 13, 25, 29, 30, 31, 32], "partit": [1, 4, 9], "partli": [6, 28], "partner": 15, "pass": [2, 3, 12, 14, 31], "password": 23, "past": [10, 25, 31], "patch": [6, 25, 32], "path": [0, 4, 6, 7, 9, 21, 28, 31, 32], "pathcollect": 17, "patholog": [], "patient": 7, "patter": 4, "pattern": [0, 3, 4, 12, 27, 28, 31], "paul": [], "pauli": [0, 28], "pav": [], "pc": [11, 15, 21], "pca": [0, 7, 21, 28, 29], "pd": [0, 4, 5, 6, 7, 9, 11, 28, 29, 30, 31, 32], "pde": 2, "pdf": [0, 3, 4, 5, 6, 9, 15, 16, 19, 20, 23, 27, 28, 32], "pedagog": [0, 28, 29], "penal": [6, 18, 29, 31], "penalti": [6, 13, 18, 23, 29, 31], "penros": [5, 6], "pentagon": [13, 30], "peopl": [1, 9, 13, 21, 31], "per": [0, 1, 6, 24, 26, 28, 31, 32], "percentag": [10, 11, 26], "perceptron": [0, 1, 7, 28], "peregrin": 28, "perez": [], "perfect": [0, 1, 13, 28, 31], "perfectli": [4, 6, 32], "perform": [0, 2, 3, 4, 5, 6, 8, 10, 11, 12, 13, 14, 16, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "performac": 4, "perhap": [0, 5, 13, 28, 29, 30, 31], "perimet": 1, "period": [1, 4, 25], "permiss": 15, "permit": [], "permut": 11, "persist": 13, "person": [5, 6, 7, 16, 20, 24, 26, 28, 29], "perspect": 27, "pertin": [12, 28], "petal": [8, 9], "peter": [27, 29], "phantom": 25, "phase": [6, 12], "phenomena": 25, "phenomenon": 31, "phi": 8, "phi_k": 8, "philipp": [], "philosophi": 13, "phone": [26, 28], "photo": [4, 28], "php": 23, "phrase": [0, 28], "physic": [0, 1, 4, 7, 12, 13, 25, 26, 27, 28, 29, 30, 31, 32], "pi": [2, 3, 5, 6, 7, 9, 12, 13, 25, 32], "pick": [1, 9, 10, 11, 13, 14, 31], "pickl": 1, "pictur": [0, 28], "pie": [21, 28], "piec": [11, 14], "pierr": [], "pillow": [0, 21, 23, 28], "pinv": [5, 6, 13, 23, 29, 30, 31], "pip": [0, 1, 15, 21, 23, 28], "pip3": [0, 1, 23, 28], "pipelin": [0, 6, 8, 10, 29, 32], "pippin": 28, "pit": 4, "pitfal": [6, 29], "pitt": 12, "pixel": [1, 3, 4, 28], "pixel_height": [1, 3], "pixel_width": [1, 3], "pkg_resourc": [], "pkgutil": [], "place": [0, 4, 6, 8, 13, 15, 22, 23, 28, 30, 32], "plai": [0, 3, 4, 5, 6, 8, 11, 18, 21, 23, 28, 29, 30, 32], "plain": [8, 10, 12, 13, 14, 30, 31], "plan": [6, 9, 26, 27, 28], "plane": [8, 9], "plateau": [5, 30, 31], "platform": [21, 28], "plausibl": 12, "pleas": [13, 23, 26, 28], "plenti": 1, "plethora": [3, 12], "plot": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31], "plot_all_sc": [23, 29], "plot_confusion_matrix": [7, 10], "plot_count": 6, "plot_cumulative_gain": [7, 10], "plot_data": 1, "plot_dataset": 8, "plot_decision_boundari": [9, 10], "plot_import": 10, "plot_max": 4, "plot_min": 4, "plot_model": 4, "plot_numb": 4, "plot_predict": 8, "plot_regression_predict": 9, "plot_result": 4, "plot_roc": [7, 10], "plot_surfac": [2, 6, 13], "plot_train": 9, "plot_tre": [9, 10], "plt": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 22, 25, 28, 29, 30, 31, 32], "plu": [0, 3, 5, 7, 18, 28, 29], "plugin": [], "pm": [8, 32], "pmatrix": 2, "pml": 27, "pn": 3, "png": [0, 4, 6, 7, 9, 28, 32], "point": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 18, 19, 20, 22, 23, 25, 26, 28, 29, 30, 31, 32], "point_1": 4, "point_2": 4, "poisson": [21, 25, 28], "poli": [6, 8, 32], "poly100_kernel_svm_clf": 8, "poly3": 0, "poly3_plot": 0, "poly_featur": [8, 9, 15], "poly_features10": 9, "poly_fit": 9, "poly_fit10": 9, "poly_kernel_svm_clf": 8, "poly_model": 15, "poly_ms": 15, "poly_predict": 15, "polydegre": [0, 5, 6, 10, 29, 32], "polygon": [13, 30], "polym": 12, "polymi": 23, "polynomi": [0, 5, 6, 7, 8, 9, 10, 11, 15, 17, 19, 20, 23, 28, 29, 31, 32], "polynomial_featur": [6, 15, 16, 17, 32], "polynomial_svm_clf": 8, "polynomialfeatur": [0, 6, 8, 9, 15, 16, 19, 29, 32], "polytrop": [0, 6, 32], "pool": 3, "pool_siz": 3, "poor": [1, 13, 30, 31], "poorli": [0, 29], "popul": [0, 5, 28, 29], "popular": [0, 1, 3, 6, 7, 8, 9, 11, 12, 15, 21, 22, 23, 25, 29], "popularli": [0, 28], "portabl": 10, "portion": [11, 13, 31], "pose": [0, 4, 5, 6, 11, 25, 28, 32], "posit": [0, 1, 2, 3, 5, 7, 8, 10, 11, 13, 14, 22, 25, 28, 29, 30, 31], "possibl": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "possibli": [6, 8, 13, 23], "post": [], "posterior": 5, "postpon": [0, 29], "postscript": 23, "postul": 5, "potenti": [0, 3, 5, 6, 12, 13, 29, 31, 32], "pott": 12, "power": [0, 1, 5, 6, 8, 9, 12, 13, 28, 29, 30, 31, 32], "pp": [5, 6, 19, 32], "practic": [0, 5, 6, 7, 8, 16, 18, 19, 23, 25, 29, 32], "practition": [0, 1, 3, 28, 31], "pre": 28, "preambl": [], "preced": [1, 11, 12, 25], "preceed": 4, "preceq": 8, "precis": [0, 2, 5, 11, 13, 22, 23, 25, 28, 29, 31, 32], "pred": [6, 32], "predicit": 0, "predict": [0, 1, 5, 6, 7, 8, 9, 10, 15, 16, 17, 19, 21, 23, 27, 28, 29, 30, 31, 32], "predict_prob": 1, "predict_proba": [7, 10], "predictor": [0, 5, 6, 7, 9, 10, 11, 28, 29, 31], "prefer": [0, 1, 6, 8, 9, 11, 13, 15, 20, 21, 23, 28], "prefil": [], "prepar": [0, 6, 22, 23, 28, 29], "preprocess": [0, 4, 6, 7, 8, 9, 10, 11, 15, 16, 17, 18, 19, 23, 32], "prerequisit": 0, "prescript": 23, "presenc": 13, "present": [0, 5, 6, 7, 9, 12, 13, 22, 23, 25, 28, 29, 30, 31], "preserv": [3, 11, 22], "press": [13, 15, 27, 30], "pretrain": [1, 4], "pretti": [0, 4, 8, 9, 21, 23, 28], "prettier": [], "prev_centroid": 14, "prevent": [13, 25, 31], "previou": [0, 1, 2, 3, 4, 5, 6, 8, 10, 11, 12, 13, 15, 16, 22, 23, 25, 29, 30, 31], "previous": [2, 3, 9, 10, 25], "price": [0, 4, 9, 13, 31], "primal": 8, "primari": [0, 7, 28], "prime": 25, "princip": [0, 5, 7, 21, 28, 29, 30], "principl": [0, 6, 7, 8, 14, 28, 32], "print": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 18, 22, 25, 28, 29, 30, 31, 32], "print_funct": [8, 9], "printout": [0, 28], "prior": [0, 5, 6, 28], "privat": 0, "prob": [1, 25], "probabilist": [0, 27, 28, 29], "probabl": [0, 1, 3, 4, 6, 7, 10, 13, 21, 28, 29, 31], "problem": [0, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 17, 19, 21, 22, 23, 25, 32], "probml": 27, "proce": [0, 5, 6, 7, 8, 9, 10, 11, 13, 22, 28, 29, 32], "procedur": [2, 4, 5, 6, 8, 10, 11, 13, 29, 30, 31, 32], "proceed": 22, "process": [0, 2, 4, 6, 9, 10, 12, 13, 21, 22, 23, 25, 27, 28, 30, 31, 32], "procur": [], "prod": 27, "prod_": [1, 5, 7, 32], "produc": [0, 3, 4, 5, 6, 9, 10, 11, 12, 13, 18, 20, 21, 22, 25, 28, 29, 32], "product": [0, 1, 3, 5, 6, 7, 8, 12, 13, 16, 17, 21, 22, 28, 29, 31, 32], "profess": [0, 28], "profit": [], "program": [0, 1, 4, 5, 6, 8, 12, 14, 15, 21, 22, 24, 25, 26, 28, 29], "programm": 22, "progress": [1, 4, 14, 31], "prohibit": [6, 32], "project": [0, 1, 2, 3, 5, 11, 13, 15, 19, 21, 24, 29, 30, 31, 32], "project_root_dir": [0, 6, 7, 9, 28, 32], "promin": 12, "promis": 8, "promot": [26, 28], "prompt": 20, "prone": [9, 15], "pronounc": [13, 21, 28, 31], "proof": [0, 11, 12, 13, 28, 30, 32], "prop": 31, "prop_cycl": [], "propag": [2, 3, 13, 31], "proper": [0, 2, 6, 7, 20, 32], "properli": [1, 6, 8, 10, 13, 18, 20, 23, 31], "properti": [0, 1, 3, 12, 13, 16, 22, 28, 32], "proport": [0, 1, 5, 9, 11, 13, 25, 28, 29], "propos": [1, 4, 6, 10, 23, 28, 31], "propto": [5, 13, 30, 31], "proton": [0, 28], "prove": [3, 13, 30, 31], "provid": [0, 1, 3, 4, 5, 6, 8, 9, 10, 12, 13, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "proxi": [1, 13, 31], "prune": 9, "pseudo": [22, 25, 31], "pseudocod": 23, "pseudoinv": 5, "pseudoinvers": [5, 6, 23], "pseudorandom": [6, 25, 32], "psychologi": [0, 28], "pt": 13, "public": [0, 15, 21, 28], "publish": [], "pull": 15, "punish": [0, 1, 28], "pure": [3, 9, 25], "purest": 9, "puriti": 9, "purpos": [0, 3, 10, 12, 14, 28], "push": 15, "put": [1, 20, 31], "putmask": [], "py": 5, "pybtex": [], "pycod": 28, "pydata": 21, "pydevd_extension_api": [], "pydevd_plugin": [], "pydevd_plugin_plugin_nam": [], "pydot": 9, "pygment": [], "pyhton2": 28, "pylab": [7, 28], "pypi": 21, "pyplot": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 22, 25, 28, 29, 30, 31, 32], "pythagora": 5, "python": [1, 2, 3, 5, 6, 8, 11, 12, 13, 14, 18, 20, 23, 25, 29, 31], "python2": [0, 23], "python3": [0, 21, 23, 28], "pythonpath": [], "pytorch": [0, 21, 23, 28], "pyzmq": [], "q": [5, 6, 8, 11, 25, 32], "qp": 8, "qquad": [2, 11, 13, 22, 31], "qr": [5, 6, 22, 29, 30], "quad": [1, 13, 22], "quadrat": [0, 8, 9, 13, 28], "qualit": [4, 9, 23, 25], "qualiti": [0, 9, 21, 28, 29], "quantifi": 1, "quantil": 10, "quantit": [0, 6, 9, 23, 28, 32], "quantiti": [0, 2, 5, 6, 7, 9, 10, 11, 12, 14, 16, 22, 25, 28, 29, 30, 31, 32], "quantum": [4, 12, 27, 28], "quartil": [0, 29, 31], "quench": 5, "queri": 9, "question": [0, 5, 6, 9, 11, 12, 13, 23, 26, 28, 29, 31, 32], "qugan": 4, "quick": [4, 25], "quicker": 31, "quickli": [1, 3, 9, 11, 13, 30, 31], "quit": [1, 5, 6, 9, 10, 12, 15, 29, 30, 32], "quot": 4, "r": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 21, 22, 23, 25, 29, 30, 31, 32], "r2": [0, 5, 6, 19, 28, 29, 30], "r2_score": [0, 28], "r2score": [0, 28], "r_": 31, "r_0": 31, "r_1": 9, "r_2": 9, "r_j": 9, "r_m": 9, "r_t": 31, "rad": [], "rade": [], "radial": [8, 12], "radioact": 25, "radiu": [0, 1, 29, 31], "radziej": [], "ragan": [], "rain": 9, "rais": [], "ram": 31, "ramanujam": [], "ramp": 1, "ran0": 25, "ran1": 25, "ran2": 25, "ran3": 25, "rand": [0, 4, 5, 6, 9, 10, 13, 15, 19, 22, 28, 29, 30, 31, 32], "randint": [6, 9, 13, 31, 32], "randn": [0, 1, 2, 5, 6, 9, 11, 13, 15, 18, 28, 29, 30, 31, 32], "random": [0, 1, 2, 3, 4, 5, 6, 8, 9, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 28, 29, 30, 31, 32], "random_forest_model": 10, "random_index": [13, 31], "random_indic": [1, 3], "random_st": [7, 8, 9, 10, 11], "randomforestclassifi": 10, "randomli": [1, 6, 9, 13, 14, 18, 30, 31, 32], "rang": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 18, 19, 22, 25, 28, 29, 30, 31, 32], "rangl": [0, 6, 11, 25, 28, 29], "rangle_x": 25, "rank": [5, 29, 30], "rankdir": 4, "raphson": [1, 8, 13], "rapidli": [0, 31], "rare": [1, 13, 31], "raschka": [28, 29, 32], "rasckha": 28, "rashcka": [30, 31], "rate": [1, 2, 3, 4, 8, 9, 10, 12, 13, 18, 30, 32], "rather": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 22, 25, 28, 29, 30, 32], "ratio": [4, 7, 9, 10, 11], "rational": [0, 28], "ravel": [5, 6, 7, 8, 9, 10, 11, 13, 22, 32], "raw": [3, 31], "rbf": [8, 11, 12], "rbf_kernel_svm_clf": 8, "rbf_pca": 11, "rc": 25, "rcond": [0, 28, 29], "rcparam": [1, 3, 7, 8, 9, 10, 25, 28], "re": [2, 4, 13, 15, 30], "reach": [1, 4, 5, 6, 9, 10, 12, 13, 14, 30, 31, 32], "react": [], "read": [0, 2, 3, 4, 5, 6, 7, 8, 11, 12, 16, 17, 19, 20, 22, 23, 25, 27, 30], "read_csv": [0, 6, 7, 9, 32], "read_fwf": [0, 28], "reader": [0, 6, 20, 22, 25, 28, 29, 31], "readi": [0, 1, 5, 6, 8, 10, 11, 12, 22, 28], "readili": 1, "readm": [15, 20], "readthedoc": 21, "real": [0, 1, 4, 7, 10, 11, 12, 16, 18, 19, 22, 29, 32], "real_loss": 4, "real_output": 4, "realist": [8, 28], "realiti": 25, "realiz": [1, 12], "realli": [0, 1, 28], "rearrang": 13, "reason": [0, 1, 3, 4, 10, 13, 27, 28, 30, 31], "reassign": 1, "recal": [5, 6, 9, 10, 11, 12, 22, 25, 28, 29, 30, 31, 32], "recarrai": [], "recast": 3, "receiv": [1, 3, 10, 12, 25], "recent": [0, 6, 13, 27, 31, 32], "recept": [3, 12], "receptive_field": 3, "recip": [0, 6, 7, 22, 23, 28, 29], "reciproc": 5, "recogn": [0, 4, 5, 10, 28, 32], "recognit": [0, 1, 3, 12, 27, 28], "recommen": 28, "recommend": [0, 2, 3, 4, 5, 6, 8, 13, 15, 19, 20, 21, 22, 23, 27, 30, 31, 32], "reconsid": 9, "reconstruct": 11, "record": [10, 23, 24, 26, 28], "recreat": 15, "rectangl": [9, 13, 30], "rectangular": [5, 29, 30], "rectifi": [1, 3, 12], "recur": [0, 21, 28], "recurr": [0, 1, 21, 28], "recurs": [9, 21, 22, 28], "red": [0, 3, 4, 6, 8, 9, 31, 32], "redefin": [0, 10, 28, 29, 30], "redefinit": 30, "redistribut": [], "reduc": [1, 3, 5, 6, 9, 10, 11, 13, 28, 30, 31, 32], "reduct": [0, 10, 11, 21, 25, 28, 29], "reegress": 23, "ref": 20, "refer": [0, 1, 2, 3, 5, 6, 11, 12, 13, 14, 20, 22, 27, 28, 29, 30, 31, 32], "referansestil": 20, "referenc": 2, "refin": 12, "refit": [6, 32], "reflect": [0, 1, 4, 5, 23, 25, 28], "refresh": [21, 28], "refreshprogrammingskil": 28, "reg": [10, 11], "regard": [1, 9, 13], "regardless": [12, 16], "regexp": [], "reggi": [], "regim": 31, "region": [3, 4, 6, 9, 12, 23, 31], "regist": [6, 25], "reglasso": [5, 30], "regr_1": [0, 9], "regr_2": [0, 9], "regr_3": [0, 9], "regress": [1, 8, 11, 12, 16, 20, 21, 22], "regressor": [0, 7, 10], "regret": [], "regridg": [0, 5, 6, 29, 30, 31], "regular": [0, 3, 4, 5, 6, 7, 9, 13, 17, 18, 26, 28, 29, 30, 31, 32], "regularli": 15, "reilli": [0, 27, 28], "reinforc": [0, 8, 21, 28], "reiter": 1, "reitz": [], "reject": 7, "rel": [0, 4, 6, 7, 9, 12, 13, 25, 28, 29, 31, 32], "relat": [0, 1, 3, 4, 5, 11, 13, 14, 19, 22, 25, 28, 29, 30, 32], "relationship": [0, 4, 9, 18, 28], "relativeerror": [0, 28, 29], "releas": [1, 21, 28], "relev": [0, 1, 5, 7, 11, 21, 23, 25, 28, 30, 31], "reli": [0, 6, 8, 31], "reliabilti": 23, "reliabl": [7, 25], "relu": [3, 4, 28], "remain": [1, 2, 4, 6, 12, 22, 25, 29, 31, 32], "remaind": 25, "reman": 2, "remark": 1, "rememb": [0, 8, 13, 20, 22, 23, 28, 31], "remind": [0, 5, 11, 13, 19, 22, 25, 32], "remot": 15, "remov": [4, 5, 6, 18, 29, 30, 31], "renam": 15, "render": [0, 28, 29], "reorder": [5, 7, 29, 30], "reorgan": [0, 28], "repeat": [0, 1, 3, 4, 5, 6, 9, 10, 11, 13, 14, 22, 23, 25, 28, 29, 30, 31, 32], "repeated": 28, "repeatedli": [0, 6, 10, 13, 32], "repet": 3, "repetit": [6, 28, 29, 32], "rephras": [13, 30], "replac": [0, 1, 3, 4, 5, 6, 10, 12, 14, 21, 23, 28, 29, 30, 32], "replica": [6, 32], "repo": [15, 23], "report": [28, 31], "repositori": [4, 20, 23, 28], "reposotori": [], "repres": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 23, 25, 28, 29, 30, 31, 32], "represent": [0, 1, 3, 6, 25, 28, 32], "representd": 3, "reproduc": [0, 5, 6, 9, 12, 15, 16, 18, 20, 21, 25, 28, 29], "repuls": [0, 28], "request": [0, 13, 31], "requir": [0, 1, 3, 4, 5, 6, 8, 9, 11, 12, 13, 15, 17, 18, 19, 20, 22, 28, 29, 30, 31, 32], "res1": 2, "res2": 2, "res3": 2, "res_analyt": 2, "res_analytical1": 2, "res_analytical2": 2, "res_analytical3": 2, "resaml": 6, "resampl": [0, 7, 10, 21, 28, 29], "rescal": [0, 11, 12, 31], "rescu": 5, "reseach": 6, "research": [0, 4, 13, 21, 27, 28, 31], "resembl": [6, 25, 32], "reserv": [1, 5, 6, 25, 32], "reshap": [0, 1, 2, 3, 4, 6, 8, 9, 10, 22, 28, 29, 32], "resid": 31, "residenti": [], "residu": [0, 5, 13, 28], "resiz": [5, 29, 30], "resnet": 31, "resort": 31, "resourc": [28, 31], "respect": [0, 1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 13, 14, 16, 17, 18, 23, 25, 28, 29, 30, 31, 32], "respond": 12, "respons": [0, 7, 9, 12, 28, 29], "rest": [0, 5, 18, 29, 30, 31], "restat": [0, 12, 28], "restor": 4, "restored_discrimin": 4, "restored_gener": 4, "restrict": [0, 3, 9, 12, 28], "result": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 28, 31, 32], "retail": [], "retain": [5, 6, 29, 30, 31, 32], "rethink": 32, "return": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 13, 14, 16, 17, 22, 25, 28, 29, 30, 31, 32], "return_data": 14, "return_sequ": 4, "return_x_i": 9, "reus": [1, 3, 6, 19, 20, 23], "reveal": [0, 12, 28], "revers": [1, 22], "review": [21, 22], "revis": [], "revisit": 14, "revolut": 28, "reward": [0, 4, 28], "rewrit": [0, 3, 5, 6, 7, 8, 10, 11, 12, 13, 16, 19, 22, 23, 25, 30, 31], "rewritten": [2, 6, 8, 10, 25, 32], "rewrot": 13, "rf": 10, "rgb": 3, "rgoj5yh7evk": 21, "rh": [6, 32], "rho": [0, 10, 13, 31], "rho_1": 10, "rho_2": 10, "rho_m": 10, "rich": [0, 28], "rid": [], "ride": 9, "rideclass": 9, "ridedata": 9, "ridg": [7, 11, 13, 20, 21, 28, 32], "ridge_paramet": 17, "ridge_sk": 6, "ridgebeta": 30, "ridgetheta": 5, "right": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 14, 16, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "right_sid": 2, "rightarrow": [0, 1, 5, 6, 8, 11, 12, 13, 25, 28, 29, 30, 31, 32], "rigor": [0, 28, 29, 30], "ring": 6, "rise": [0, 28], "risk": [0, 13, 28, 30, 31], "rival": 4, "river": [], "rlm": 28, "rm": [25, 31], "rmse": [], "rmsporp": [13, 31], "rmsprop": [1, 3, 4, 13, 23, 32], "rnd_clf": 10, "rng": 25, "rnn": [4, 12], "rnn1": 4, "rnn2": 4, "rnn_2layer": 4, "rnn_input": 4, "rnn_output": 4, "rnn_train": 4, "rntrick1": 25, "rntrick2": 25, "rntrick3": 25, "rntrick4": 25, "ro": [0, 13, 28, 30, 31], "robert": [19, 23, 27], "robust": [0, 28, 31], "robustscal": [0, 29, 31], "roc": [7, 10], "role": [0, 2, 5, 6, 8, 18, 21, 23, 28, 29, 30, 31, 32], "roll": 6, "ronach": [], "room": [0, 26, 28], "root": [0, 5, 9, 13, 15, 25, 29, 30, 31], "root_directori": [], "rot": 28, "rotat": [1, 8, 9, 10], "rotation_matrix": 9, "roughli": [1, 3, 18], "round": [7, 9, 13], "routin": [13, 22, 28, 30], "row": [0, 1, 2, 5, 6, 9, 11, 16, 22, 28, 29, 30, 32], "rr": [5, 29, 30], "rrr": [5, 29, 30], "rubric": [], "rudg": [], "rug": [13, 30, 31], "rule": [0, 1, 5, 6, 13, 23, 28, 29, 30], "run": [0, 1, 2, 4, 5, 6, 8, 9, 11, 13, 15, 20, 21, 23, 28, 29, 30, 31, 32], "runtim": [1, 6, 14, 15], "rust": [0, 21, 22, 28], "rvert": 1, "rvert_2": 1, "s_": [3, 6], "s_1": 6, "s_i": [6, 7], "s_j": 6, "s_k": 6, "s_phenomenon": 23, "saddl": [13, 30, 31], "safeguard": [18, 31], "sai": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 19, 22, 23, 25, 28, 29, 30, 31, 32], "said": [6, 9, 13, 30], "sake": [0, 5, 7, 11, 28, 29, 30], "sale": [0, 28], "sam": 28, "same": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 14, 15, 16, 18, 20, 22, 23, 25, 28, 29, 30], "samm": 10, "sampl": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 14, 18, 19, 21, 22, 23, 25, 28, 29, 31, 32], "sample_vari": 14, "sampleexptvari": 25, "samwis": 28, "sandbox": [], "sasha": [], "sastri": 11, "satisfactori": [0, 28], "satisfi": [1, 2, 3, 6, 8, 13, 22, 25, 30, 32], "satur": [1, 6, 32], "save": [0, 4, 6, 7, 9, 13, 20, 28, 31, 32], "save_fig": [0, 6, 7, 9, 10, 28, 32], "savefig": [0, 4, 6, 7, 9, 25, 28, 32], "savetxt": 4, "saw": [5, 29], "scalabl": 10, "scalar": [2, 5, 6, 10, 29, 32], "scale": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 21, 22, 23, 26, 28, 30], "scale_mean": 4, "scale_std": 4, "scaler": [0, 7, 8, 9, 10, 11, 17, 23, 29], "scan": [5, 7], "scari": 5, "scatter": [0, 1, 6, 7, 8, 9, 14, 15, 17, 28, 29, 31, 32], "scenario": [6, 13, 30, 31], "schedul": [13, 31], "scheme": [1, 13, 30, 31], "schrage": 25, "sch\u00f8yen": [6, 29, 31], "scienc": [0, 1, 10, 12, 13, 21, 24, 25, 26, 27, 30, 32], "scientif": [0, 20, 21, 23, 28], "scientist": [0, 28], "scikit": [3, 5, 6, 8, 9, 10, 13, 15, 16, 20, 21, 22, 23, 27], "scikit_learn": 0, "scikitlearn": 28, "scikitplot": [7, 10], "scipi": [0, 3, 5, 6, 13, 21, 22, 23, 28, 29, 30, 32], "scl": 6, "scm": 15, "score": [0, 1, 3, 6, 7, 9, 10, 11, 15, 16, 19, 23, 26, 28, 29, 31, 32], "scores_kfold": [6, 32], "scratch": [1, 13, 16], "script": [], "sdg": [13, 31], "sdv4f4s2sb8": [30, 31], "seaborn": [0, 1, 3, 6, 7, 28], "seamless": [0, 21, 23, 28], "search": [0, 1, 3, 5, 9, 13, 15, 28, 30, 31], "sebastian": 28, "sebastianraschka": 28, "sec": 6, "second": [0, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 14, 15, 16, 20, 21, 22, 25, 26, 28, 29, 30, 32], "second_mo": 31, "second_term": 31, "secondari": 31, "secondeigvector": 11, "secondli": 12, "section": [4, 11, 16, 20, 22, 23, 25, 29, 31], "sector": 0, "see": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 15, 16, 18, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "seed": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 13, 14, 18, 20, 25, 28, 29, 30, 31, 32], "seed_imag": 4, "seek": [1, 2, 8], "seem": [1, 3, 4, 31], "seemingli": [0, 28], "seen": [0, 1, 3, 5, 10, 12, 25], "segment": [13, 30], "seismic": 6, "seldomli": [0, 28], "select": [1, 5, 6, 8, 9, 10, 11, 15, 20, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32], "selevet": 15, "self": [1, 5, 29], "sell": 4, "semest": [7, 24], "semi": [8, 13, 30, 31], "semilogx": 6, "send": [5, 12, 13, 26, 28], "senior": [24, 26], "sens": [0, 4, 6, 8, 28, 32], "sensibl": 3, "sensit": [0, 5, 6, 9, 13, 28, 29, 31, 32], "sent": 2, "sentenc": [4, 12], "separ": [0, 1, 2, 4, 6, 8, 9, 12, 14, 18, 21, 23, 25, 28, 31, 32], "septemb": [18, 23, 28], "sequenc": [3, 4, 7, 9, 10, 12, 13, 21, 22, 25, 28, 30], "sequenti": [1, 3, 4, 10, 12, 25], "seri": [0, 1, 2, 3, 4, 5, 6, 10, 11, 12, 13, 22, 28, 29, 30, 32], "serif": [7, 25, 28], "serv": [0, 1, 2, 3, 5, 7, 13, 27, 28, 29, 30, 31], "servic": 23, "session": [1, 15, 20, 23, 24, 26, 28], "set": [1, 4, 5, 6, 7, 8, 10, 11, 13, 14, 16, 17, 18, 21, 22, 23, 25, 26, 31, 32], "set_major_formatt": 6, "set_major_loc": 6, "set_tick": [1, 8], "set_ticklabel": 1, "set_titl": [0, 1, 2, 3, 7, 12, 14, 28], "set_xlabel": [0, 1, 2, 3, 7, 12, 28], "set_xlim": [7, 12], "set_xticklabel": 1, "set_ylabel": [0, 1, 2, 3, 7, 28], "set_ylim": [7, 12], "set_ytick": 7, "set_yticklabel": [1, 6], "set_zlim": 6, "seth": 4, "setminu": 6, "setosa": [8, 9], "setosa_or_versicolor": 8, "setp": [6, 32], "setup": [1, 4, 6, 8, 21, 28, 29, 30], "sever": [0, 3, 5, 6, 7, 8, 9, 11, 12, 13, 16, 21, 22, 23, 25, 28, 29, 30, 31, 32], "sgd": [1, 3, 30], "sgd_clf": 8, "sgdclassifi": 8, "sgdreg": 13, "sgdregressor": 13, "sgn": [5, 29, 30], "shall": [], "shallow": [13, 31], "shape": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 18, 22, 28, 29, 30, 31, 32], "share": [1, 3, 15, 28], "share_mask": [], "shareabl": 15, "she": 7, "sheppard": [], "shibukawa": [], "shift": [1, 6, 12, 15, 18, 25, 29, 31], "ship": 3, "shire": 28, "short": [4, 5, 20, 23], "shortcom": [13, 30, 31], "shorten": 4, "shorter": 25, "shorthand": [28, 32], "shortli": [22, 28], "should": [0, 2, 3, 5, 6, 8, 9, 11, 12, 15, 18, 19, 20, 22, 23, 25, 28, 29, 31, 32], "shouldn": [], "show": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "show_shap": 4, "shown": [0, 4, 5, 8, 12, 13, 22, 29, 30, 31], "shrink": [3, 5, 6, 8, 11, 29, 30, 31], "shrinkag": [5, 6, 29, 30], "shrunk": 11, "shuffl": [0, 1, 4, 6, 13, 29, 31, 32], "side": [0, 2, 5, 8, 12, 13, 22, 23, 28, 30], "sigh": [21, 28], "sigma": [0, 1, 5, 6, 7, 10, 11, 12, 13, 19, 22, 23, 25, 28, 29, 30, 31, 32], "sigma0": 25, "sigma1": 25, "sigma2": 25, "sigma_": [5, 22, 28, 29, 30, 32], "sigma_0": [5, 29, 30], "sigma_1": [5, 29, 30], "sigma_2": [5, 29, 30], "sigma_fn": [7, 12], "sigma_i": [0, 5, 28, 29, 30], "sigma_j": [5, 29, 30], "sigma_m": [6, 25, 32], "sigma_n": [11, 25], "sigma_t": 13, "sigma_x": 25, "sigmoid": [1, 2, 4, 7, 8, 10, 12], "sigmundson": [6, 29, 31], "sign": [1, 2, 7, 8, 10, 25, 26], "signal": [1, 3, 10, 12, 31], "signifi": 4, "signific": [1, 31], "significantli": [1, 13, 18, 25, 30, 31], "sim": [4, 5, 6, 13, 19, 25, 32], "similar": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 14, 18, 21, 22, 23, 28, 30, 32], "similarli": [0, 1, 3, 5, 8, 10, 13, 25, 28, 29, 30, 31], "simpl": [1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 14, 16, 17, 21, 22, 25, 32], "simple_plot": [], "simplepredict": 10, "simpler": [0, 1, 5, 6, 7, 13, 16, 21, 23, 28, 30, 31], "simplernn": 4, "simplest": [0, 1, 3, 4, 9, 10, 12, 14, 23, 28], "simpletre": 10, "simpli": [0, 1, 2, 4, 5, 6, 8, 9, 10, 11, 12, 21, 22, 23, 25, 28, 29, 30, 31, 32], "simplic": [2, 5, 6, 7, 8, 9, 10, 11, 12, 14, 29, 30, 31], "simplicti": [5, 29, 30], "simplifi": [0, 6, 9, 18, 21, 23, 28, 29, 31, 32], "simplist": [3, 6, 25, 32], "simul": [6, 18, 31, 32], "simultan": [6, 31, 32], "sin": [0, 1, 2, 3, 4, 9, 12, 13, 22, 28], "sinc": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 16, 18, 22, 23, 25, 27, 28, 29, 30, 31, 32], "sine": [3, 12], "singl": [0, 1, 2, 3, 5, 6, 7, 8, 9, 12, 13, 18, 19, 22, 25, 28, 29, 30, 31, 32], "singular": [0, 6, 13, 22, 28, 32], "sinusoid": 3, "site": [0, 23, 24, 29], "situat": [0, 4, 5, 7, 13, 25, 28, 29, 30, 31], "six": [3, 25], "size": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 13, 18, 20, 22, 23, 25, 28, 32], "sizesp": 31, "sketch": 10, "ski": 9, "skill": 0, "skip": 11, "skl": [0, 6, 28, 29, 31], "sklearn": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 17, 19, 20, 28, 29, 30, 31, 32], "skplt": [7, 10], "sl": [6, 29, 31], "slack": 8, "slender": [], "slice": [2, 22, 28], "slide": [0, 3, 16, 23, 25, 28, 29, 30], "slight": [6, 13, 32], "slightli": [1, 2, 3, 5, 6, 7, 10, 25, 29, 30, 32], "slope": [8, 11, 12], "slow": [0, 2, 8, 13, 18, 29, 30, 31], "slower": [5, 22, 28, 29, 30, 31], "slowest": 22, "slowli": [12, 31], "slp": 1, "small": [0, 1, 2, 3, 5, 6, 8, 9, 10, 11, 12, 13, 18, 21, 22, 25, 28, 29, 30, 31, 32], "smaller": [0, 1, 2, 5, 6, 8, 9, 11, 13, 25, 28, 29, 30, 31, 32], "smallest": [0, 4, 14, 28], "smallest_row_index": 14, "smodin": [], "smooth": [0, 3, 6, 13, 23, 28, 30, 31], "smoother": 31, "sn": [0, 1, 3, 6, 7, 28], "sne": 11, "so": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "soar": 6, "social": 0, "soft": [1, 7, 10, 12], "soften": 8, "softmax": [3, 7], "softwar": [0, 8, 21, 22], "sokogskriv": 20, "sol": 8, "sole": [0, 6, 28], "solid": [0, 7], "solut": [0, 1, 2, 3, 5, 6, 8, 10, 11, 13, 18, 22, 23, 25, 28, 29, 30, 31, 32], "solution_ev": 31, "soluton": 2, "solv": [0, 1, 3, 5, 6, 8, 10, 11, 12, 13, 16, 22, 23, 28, 29], "solve_expdec": 2, "solve_ode_deep_neural_network": 2, "solve_ode_neural_network": 2, "solve_pde_deep_neural_network": 2, "solveod": 2, "solveode_popul": 2, "solver": [2, 7, 8, 9, 10, 22, 28], "some": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 15, 16, 18, 19, 23, 25, 28, 31, 32], "some_model": [6, 29, 31], "somehow": 4, "someon": 16, "someth": [0, 1, 3, 4, 7, 9, 11, 15, 19, 20, 23, 25, 28, 29], "sometim": [0, 1, 11, 12, 13, 14, 19, 29, 31], "soon": [22, 26, 29], "sophist": [0, 28], "sopt": 13, "sort": [5, 6, 9, 11, 25, 32], "sound": [3, 5], "sourc": [0, 1, 3, 6, 21, 22, 23, 25, 28, 31, 32], "space": [0, 1, 4, 5, 8, 9, 11, 12, 13, 14, 25, 29, 30, 31], "span": [0, 3, 5, 9, 11, 22, 28, 29, 30], "spare": 1, "spars": [3, 6, 18, 22, 28, 31], "sparse_mtx": [22, 28], "sparsecategoricalcrossentropi": 3, "sparsiti": [10, 18], "spatial": [1, 2, 3, 12], "speak": 25, "special": [6, 7, 10, 12, 13, 22, 25, 28, 29, 30, 31], "specif": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 15, 16, 21, 22, 23, 25, 27, 28, 29, 30, 32], "specifi": [0, 3, 5, 6, 7, 9, 11, 13, 14, 25, 28, 30, 31, 32], "specifici": [0, 10, 28], "spectacular": 3, "spectral": 1, "speech": [0, 1, 3, 4, 12], "speed": [1, 2, 4, 13], "spend": [16, 25, 31], "spent": 23, "sphere": [0, 29, 31], "sphinx": [], "sphinx_book_them": [], "sphinxcontrib": [], "spike": 31, "spin": 6, "spite": 0, "spitzer": [], "spline": 8, "split": [1, 3, 4, 5, 6, 8, 9, 10, 11, 14, 16, 17, 20, 23, 25, 28, 30, 31, 32], "splite": 0, "splitter": [1, 10], "spoiler": [], "spontan": 25, "spot": 3, "spread": [0, 11, 25, 28, 29], "springer": [19, 23, 27, 28, 32], "spuriou": [13, 31], "sqquar": 30, "sqrsignal": 3, "sqrt": [3, 4, 5, 6, 8, 10, 11, 13, 25, 29, 30, 31, 32], "squar": [1, 2, 3, 4, 7, 8, 9, 11, 13, 14, 15, 17, 18, 21, 22, 25, 32], "squarederror": 10, "squaredeuclidean": 14, "squash": 12, "src": [], "srtm": 6, "srtm_data_norway_1": 6, "sso": 20, "stabil": [5, 23, 31], "stabl": [0, 4, 5, 6, 9, 16, 20, 21, 23, 28, 29, 30, 31], "stack": [3, 4], "stage": [5, 13, 15, 23, 31], "stagnat": 31, "stai": [0, 2, 4, 5, 11, 28, 29, 31], "stand": [0, 5, 9, 12, 28, 29, 30], "standard": [0, 1, 4, 5, 6, 7, 8, 10, 12, 17, 18, 19, 22, 23, 25, 28, 30, 31], "standardscal": [0, 6, 7, 8, 9, 10, 11, 17, 29, 31], "standpoint": 31, "stanford": [13, 30], "start": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 22, 25, 26, 28, 29, 30, 31, 32], "start_tim": 14, "starter": [], "stat": [6, 32], "state": [1, 2, 4, 5, 6, 7, 8, 10, 11, 12, 13, 21, 25, 28, 29, 30, 32], "statement": [0, 7, 22, 28], "static": [], "stationari": [30, 31], "statist": [0, 1, 3, 4, 7, 9, 10, 11, 12, 13, 14, 19, 22, 23, 27, 29, 30, 31], "statu": [0, 7, 15, 28], "stavang": 6, "stb": [], "std": [0, 4, 6, 18, 28, 29, 31, 32], "steep": [13, 30, 31], "steepest": 31, "stefan": [], "step": [0, 1, 2, 4, 6, 7, 9, 10, 11, 12, 13, 14, 15, 18, 22, 23, 28, 30], "step_fn": [7, 12], "step_length": [13, 31], "step_siz": 31, "steps_list": 9, "stereo": 3, "sticki": [], "still": [0, 2, 3, 5, 6, 11, 13, 25, 29, 30, 31, 32], "stimuli": 12, "stk": [27, 28], "stk2100": [27, 28], "stk3155": [15, 23, 24, 26], "stk4021": [27, 28], "stk4051": [27, 28], "stk4155": [24, 26], "stk5000": 27, "stochast": [0, 1, 5, 6, 8, 11, 12, 30, 32], "stock": 4, "stoke": 12, "stone": [0, 7], "stop": [1, 4, 9, 13, 14, 18, 30], "storag": [5, 29, 30], "store": [0, 1, 2, 3, 6, 11, 13, 25, 28, 31], "storehaug": [26, 28], "stori": [], "str": [1, 3, 4], "straight": [0, 6, 8, 13, 28, 30, 32], "straightforward": [0, 2, 3, 5, 6, 8, 9, 10, 13, 22, 28, 29, 30, 32], "strategi": [0, 1, 9, 28], "stratifi": [6, 32], "stream": 31, "strength": [0, 5, 14, 29, 30], "stretch": 11, "strict": [8, 13, 30], "strictli": [8, 13, 30], "stride": [4, 22], "strike": 6, "string": 1, "stroke": 7, "strong": [3, 6, 9, 10, 12, 22, 25, 31, 32], "strongli": [0, 8, 15, 20, 21, 22], "stronli": [], "structur": [0, 1, 2, 3, 6, 9, 10, 12, 21, 28, 32], "stuck": [1, 13, 30, 31], "student": [0, 15, 23, 24, 26, 27, 28], "studi": [0, 3, 4, 5, 6, 7, 8, 11, 12, 13, 21, 23, 27, 28, 29, 30, 31], "studier": 27, "style": [7, 9, 20, 22, 28], "stylesheet": [], "st\u00f8land": 26, "sub": [9, 12, 31], "subarrai": [], "subclass": [], "subdivid": [0, 22, 28], "subfield": 0, "subgradi": 31, "subject": [6, 8, 25], "sublicens": [], "sublinear": 31, "submit": 28, "subplot": [0, 1, 3, 4, 6, 7, 8, 9, 10, 14, 28, 32], "subplots_adjust": [8, 25], "subprogram": [22, 28], "subproject": [], "subract": [0, 29], "subroutin": [0, 28], "subscript": 1, "subsequ": [1, 4, 5, 6, 12, 22, 25, 29, 30, 32], "subset": [1, 6, 9, 12, 13, 21, 28, 30, 31, 32], "subspac": [0, 8, 11, 29], "substanti": [9, 10, 31], "substep": 11, "substitut": [3, 6, 12, 16, 22, 32], "subsubset": 9, "subtask": 6, "subtl": 1, "subtract": [0, 4, 5, 6, 11, 13, 18, 19, 22, 23, 25, 29, 31, 32], "subtre": 9, "succeed": [0, 4, 28], "success": [3, 7, 9, 13, 25], "successfulli": [4, 9], "succinctli": 31, "sudo": [0, 21, 23, 28], "suffer": [0, 1, 2, 5, 10, 28, 29, 30], "suffici": [1, 6, 8, 11, 13, 30, 32], "suggest": [1, 13, 23, 27, 30, 31], "suit": [8, 12], "suitabl": [0, 15, 19, 25, 29, 31], "sum": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 19, 22, 25, 28, 29, 30, 31], "sum_": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 19, 22, 23, 25, 28, 29, 30, 31, 32], "sum_i": [0, 2, 5, 6, 8, 13, 19, 23, 29, 30, 31, 32], "sum_j": [6, 18, 31], "sum_ja_": 0, "sum_k": [6, 8, 12, 22], "sum_logist": 13, "sum_m": 3, "sum_n": 3, "sum_nx_": 3, "summar": [5, 6, 9, 32], "summari": [1, 3, 4, 10, 24, 30, 31], "summat": [0, 3, 16, 29, 30], "sunni": 9, "super": [5, 29, 30, 31], "superfici": 3, "superscript": [1, 12], "supervis": [0, 5, 6, 7, 9, 12, 21, 28, 29, 30, 32], "supplement": [7, 23], "suppli": [], "support": [0, 1, 9, 10, 11, 13, 20, 21, 28, 29, 31], "suppos": [0, 5, 6, 7, 8, 10, 11, 12, 13, 22, 28, 29, 30, 31, 32], "suppress": [5, 13, 30], "sure": [0, 1, 4, 6, 16, 20, 23], "surf": 6, "surfac": [0, 6, 28, 31], "surpass": 6, "surpris": [0, 28], "surround": [3, 21], "survei": [0, 5, 6, 28, 29], "svc": [8, 9, 10], "svd": [0, 6, 11, 28, 32], "svdinv": 5, "svm": [8, 9, 10, 11], "svm_clf": [8, 10], "svn": [], "swath": [5, 29, 30], "switch": 0, "sy": [13, 30, 31], "symbol": [1, 5, 11, 13, 21, 25, 28, 29, 30], "symmeteri": 1, "symmetr": [0, 5, 8, 11, 12, 13, 22, 28, 29], "symmetri": 6, "sympi": [0, 21, 23, 28], "synonim": 25, "syntax": 13, "system": [0, 1, 3, 4, 6, 7, 9, 10, 12, 13, 15, 21, 22, 23, 28, 30, 31], "systemat": [4, 6, 32], "t": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 25, 26, 28, 30, 31, 32], "t0": [3, 6, 13, 31], "t1": [2, 13, 31], "t2": 2, "t3": 2, "t9jjwsmsd1o": 32, "t_": 2, "t_0": [2, 9, 13, 31], "t_1": [13, 31], "t_b": 10, "t_i": [1, 2, 5, 12, 29, 30], "t_j": 12, "t_k": 9, "tabl": [9, 23, 25, 26, 28], "tabul": [0, 28], "tabular": 28, "tackl": 4, "tag": [2, 3, 4, 5, 6, 7, 12, 13, 14, 22, 25, 29, 30], "taht": [0, 28], "tail": 25, "tailor": [2, 8, 11, 28], "taiwan": [0, 28], "take": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 17, 19, 21, 22, 25, 28, 29, 30, 31, 32], "taken": [0, 1, 3, 6, 10, 13, 22, 32], "tan": 3, "tangent": [1, 4, 12, 13, 30], "tanh": [1, 4, 7, 8, 12], "target": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 15, 16, 18, 19, 28, 29, 30, 31, 32], "target_nam": 9, "task": [0, 1, 3, 6, 9, 11, 12, 14, 23, 28, 31, 32], "tau": [3, 5, 25], "taught": 28, "tax": [], "taylor": [2, 13, 30], "taylornr": [13, 30], "tc": 8, "teach": [15, 24, 28, 32], "team": 1, "teaser": 0, "technic": [0, 5, 6, 13, 23, 30, 31, 32], "techniqu": [0, 1, 8, 10, 13, 21, 25, 27, 28, 29, 31, 32], "technologi": [0, 1], "tell": [0, 4, 6, 10, 11, 13, 16, 25, 31, 32], "temp": 1, "temp1": 1, "temp2": 1, "temperatur": [0, 9, 28], "templat": [18, 20], "temporari": [], "temporarili": 1, "ten": [3, 28], "tend": [3, 5, 6, 8, 9, 10, 12, 13, 14, 29, 31, 32], "tendenc": [0, 28], "tension": [6, 32], "tensor": 3, "tensorflow": [0, 2, 4, 8, 14, 21, 22, 23, 27, 28, 29], "term": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 19, 23, 25, 28, 29, 30, 31], "term1": [5, 6, 11], "term2": [5, 6, 11], "term3": [5, 6, 11], "term4": [5, 6, 11], "termin": [0, 4, 5, 9, 10, 13, 15, 29, 30, 31], "terminarl": 15, "terrain": 6, "terrain1": 6, "test": [3, 4, 5, 6, 7, 8, 9, 10, 13, 16, 19, 20, 22, 23, 25, 28, 30, 31, 32], "test_acc": 3, "test_accuraci": [1, 3], "test_error": 6, "test_imag": [3, 4], "test_ind": [6, 32], "test_input": 4, "test_label": [3, 4], "test_loss": 3, "test_pr": 1, "test_predict": 1, "test_rnn": 4, "test_scor": [7, 10], "test_siz": [0, 1, 3, 5, 6, 10, 15, 17, 29, 30, 31, 32], "test_split": 9, "testerror": [0, 6, 29, 32], "testi": 4, "testpredict": 4, "testx": 4, "tex": [], "text": [0, 1, 2, 4, 5, 8, 9, 11, 13, 15, 18, 20, 22, 25, 27, 29, 30, 31, 32], "textbf": [], "textbook": [16, 23, 29, 30, 32], "textual": 9, "textur": 1, "tf": [1, 3, 4, 13, 14, 30], "th": [0, 1, 2, 5, 6, 7, 9, 12, 13, 14, 22, 23, 25, 28, 29, 31, 32], "than": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 17, 21, 25, 28, 29, 31, 32], "thank": [4, 6, 29, 31], "theano": [1, 21, 28], "thei": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 16, 18, 20, 22, 23, 25, 28, 29, 30, 31, 32], "them": [0, 1, 3, 4, 6, 8, 9, 10, 11, 12, 13, 18, 22, 23, 28, 29], "theme": [0, 15, 28], "themselv": [0, 23, 25, 28, 31], "thenc": [6, 32], "theorem": [2, 6, 7, 29, 30], "theoret": [0, 4, 10], "theori": [0, 1, 3, 8, 9, 12, 13, 19, 21, 23, 27, 28, 31], "thereaft": [0, 5, 6, 11, 12, 22, 23, 28, 32], "therebi": [0, 5, 7, 11, 23, 28, 29, 30], "therefor": [0, 1, 2, 3, 4, 6, 7, 8, 11, 13, 19, 25, 28, 29, 30, 31, 32], "therein": 11, "thereof": [0, 6, 13, 28, 31, 32], "theta": [0, 1, 4, 5, 6, 7, 13, 16, 23, 25, 28, 29, 30, 31], "theta1": 31, "theta2": 31, "theta_": [0, 1, 6, 7, 13, 28, 29, 30, 31], "theta_0": [0, 5, 6, 7, 16, 28, 29, 30, 31], "theta_0x_": [0, 28, 29], "theta_1": [0, 5, 6, 7, 28, 29, 30, 31], "theta_1x_": [0, 28, 29], "theta_1x_0": [0, 28], "theta_1x_1": [0, 7, 28], "theta_1x_2": [0, 28], "theta_1x_i": [7, 29, 30, 31], "theta_2": [0, 28, 29], "theta_2x_": [0, 28, 29], "theta_2x_0": [0, 28], "theta_2x_1": [0, 28], "theta_2x_2": [0, 7, 28], "theta_2x_i": 29, "theta_3x_i": 29, "theta_4x_i": 29, "theta_closed_form": 18, "theta_closed_formol": 18, "theta_closed_formridg": 18, "theta_gdol": 18, "theta_gdridg": 18, "theta_i": [0, 1, 5, 28, 29, 30], "theta_j": [0, 5, 6, 18, 28, 29, 31], "theta_k": [30, 31], "theta_linreg": [13, 30, 31], "theta_ol": 18, "theta_p": 7, "theta_px_p": 7, "theta_ridg": 18, "theta_t": [13, 31], "theta_tru": 18, "thetaith": 31, "thetavalu": 5, "thi": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 29, 30, 31, 32], "thing": [0, 1, 2, 4, 5, 7, 9, 15, 16, 18, 25, 28, 32], "think": [0, 1, 3, 4, 6, 9, 12, 13, 14, 25, 28, 29, 30, 31, 32], "third": [0, 3, 6, 13, 26, 28, 30, 31], "thirti": 7, "thorughout": 28, "those": [0, 3, 5, 6, 8, 9, 10, 11, 22, 23, 28, 29, 30, 31, 32], "though": [1, 2, 3, 4, 13, 16, 17, 19, 22, 25, 31], "thought": [6, 14, 23, 25, 32], "thousand": [0, 1, 23, 29, 31], "three": [0, 1, 3, 5, 6, 8, 9, 12, 22, 23, 24, 25, 26, 28, 29, 30, 32], "threshold": [1, 3, 9, 10, 11, 12, 13, 31], "through": [0, 1, 2, 3, 4, 5, 6, 8, 11, 12, 13, 14, 15, 21, 22, 23, 25, 28, 29, 30, 31, 32], "throughout": [0, 4, 5, 14, 15, 21, 22, 25, 28], "throw": [3, 6, 25, 32], "thu": [0, 1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 26, 28, 29, 30, 31, 32], "thumb": [0, 6, 23, 29], "thursdai": [], "tibshirani": [6, 19, 23, 27, 28, 32], "tick_param": 6, "ticker": [6, 13, 25, 30, 31], "tif": 6, "tight_layout": [1, 7], "tightli": 11, "tild": [0, 5, 6, 7, 11, 19, 23, 25, 28, 29, 30, 31, 32], "till": [0, 4, 7, 8, 9, 10, 12, 22, 28, 29], "time": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 20, 21, 22, 23, 25, 28, 29, 30, 32], "timeit": 4, "timer": 4, "timeseri": [], "tini": [1, 31], "tip": 3, "titl": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 13, 15, 20, 25, 28, 30, 31, 32], "tm": [], "tmp": 13, "tn": [2, 3, 7], "to_categor": [1, 3, 4], "to_categorical_numpi": 1, "to_numer": [0, 6, 28, 32], "todai": 3, "togeth": [0, 3, 6, 8, 11, 13, 21, 28], "toi": 14, "token": [], "told": 13, "toler": [2, 14], "tolist": 4, "tomographi": 12, "too": [0, 2, 4, 5, 6, 9, 11, 13, 17, 18, 25, 27, 29, 30, 31, 32], "took": [8, 28], "tool": [0, 1, 3, 6, 13, 15, 21, 29, 32], "toolbox": 8, "top": [0, 3, 5, 6, 9, 10, 19, 21, 28, 32], "topic": [0, 5, 6, 7, 8, 21, 23, 29, 30, 32], "topolog": [3, 12], "topologi": [1, 12], "torkjellsdatt": [26, 28], "tort": [], "toss": [10, 25], "total": [0, 1, 2, 3, 4, 6, 7, 8, 10, 11, 12, 13, 14, 22, 25, 26, 28, 29, 30, 31, 32], "total_loss": 4, "totalclustervari": 14, "totalscatt": 14, "toward": [1, 2, 7, 12, 13, 15, 30], "towardsdatasci": 31, "town": [], "tp": [4, 7], "tpng": 9, "tpu": [13, 21, 28], "tqdm": 6, "tr": [], "track": [3, 13, 14, 15, 22, 29, 30, 31], "tract": [], "tractabl": [0, 28, 29], "trade": [5, 9, 20, 31, 32], "tradeoff": [0, 5, 19, 23, 28, 29, 30], "tradit": [0, 1, 4, 6, 28, 32], "train": [2, 3, 5, 6, 8, 9, 10, 11, 12, 13, 16, 17, 20, 23, 30, 31, 32], "train_accuraci": [0, 1, 3, 28], "train_dataset": 4, "train_end": [0, 1, 29], "train_error": 6, "train_imag": [3, 4], "train_ind": [6, 32], "train_label": [3, 4], "train_pr": 1, "train_siz": [0, 1, 3, 29], "train_step": 4, "train_test_split": [0, 1, 3, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "train_test_split_numpi": [0, 1, 29], "trainable_vari": 4, "trained_model": [6, 29, 31], "trainerror": [0, 29], "traini": 4, "training_checkpoint": 4, "training_dataset": 4, "training_gradi": [13, 31], "trainingerror": [6, 32], "trainpredict": 4, "trainscor": 4, "trainx": 4, "trait": [0, 28], "trajectori": [4, 31], "transfer": [9, 28], "transform": [0, 5, 6, 7, 8, 9, 10, 11, 12, 13, 17, 21, 22, 28, 29, 30, 31, 32], "transit": [6, 12], "translat": [1, 4, 6, 10, 28, 29, 31], "transpos": [1, 5, 11, 22, 29, 30], "travers": [0, 5], "travi": [], "treat": [0, 1, 3, 6, 12, 13, 18, 25, 28, 29, 30, 31, 32], "tree": [0, 1, 21, 28], "tree_clf": [9, 10], "tree_clf_": 9, "tree_clf_sr": 9, "tree_reg": 9, "tree_reg1": 9, "tree_reg2": 9, "trend": 25, "treue": 7, "trevor": [19, 23, 27], "tri": [2, 3, 4, 9, 13, 16, 31], "triain": 0, "trial": [0, 2, 4, 6, 13, 25, 28, 30, 31, 32], "triangl": [13, 30], "triangular": 22, "trick": [3, 4, 8, 11, 13, 25, 31], "trickier": 25, "tridiagon": 22, "trillion": 21, "trim": [], "trivial": [0, 1, 5, 11, 25, 28, 30], "troffa": [], "troubl": [0, 8, 12, 15, 29, 31], "truck": 3, "true": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "true_beta": 29, "true_fun": [6, 32], "true_theta": [6, 31], "truli": 28, "try": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 18, 21, 22, 23, 25, 28, 29, 30, 31], "tr\u00f6ger": [], "tucker": 8, "tuesdai": [26, 28], "tumor": [7, 9], "tumour": 7, "tunabl": 1, "tune": [4, 9, 13, 22, 28, 31], "turn": [0, 1, 5, 6, 7, 8, 9, 10, 11, 12, 13, 22, 23, 25, 28, 29, 30, 31, 32], "tutori": [1, 4], "tv": 2, "tveito": 2, "tweak": [1, 4, 10, 25], "twice": [13, 30], "twist": 11, "two": [0, 1, 2, 4, 5, 6, 7, 9, 10, 11, 12, 13, 15, 17, 22, 23, 24, 25, 27, 28, 29, 30, 31, 32], "tx": [13, 30, 31], "tx_1": [13, 30], "txt": [4, 15, 20], "ty": [13, 30], "type": [0, 1, 3, 6, 8, 10, 13, 22, 25, 29, 30, 31, 32], "typeset": 20, "typic": [0, 1, 2, 3, 4, 5, 7, 9, 10, 12, 13, 15, 16, 20, 25, 28, 29, 30, 31, 32], "typo": 23, "u": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 22, 23, 25, 27, 28, 29, 30, 31, 32], "u_": 22, "u_i": 12, "u_m": 10, "ua": [0, 28], "ubuntu": [0, 21, 23, 28], "uci": 23, "ufunc": [], "uio": [15, 20, 23, 26, 27], "uk": [], "un": 14, "unabl": 15, "unari": [22, 28], "unbalanc": [6, 9, 32], "unbias": [0, 5, 6, 28, 32], "uncent": [6, 29, 31], "uncertainti": [0, 5, 28], "uncertitud": 25, "unchang": [1, 3], "uncom": [], "uncorrel": [10, 25], "undefin": [5, 29, 30], "under": [0, 1, 5, 6, 10, 13, 21, 23, 28, 29, 30, 31, 32], "underdetermin": [0, 28], "underfit": [1, 6, 32], "underflowproblem": [5, 32], "undergo": 5, "undergradu": [24, 26], "underli": [0, 1, 9, 13, 18, 25, 28, 31], "underlin": [], "underscor": [], "underset": [4, 14], "understand": [0, 1, 3, 5, 6, 10, 13, 14, 15, 19, 20, 21, 28, 29, 30, 31], "understood": [8, 13], "underwai": [], "undesir": 8, "undetermin": [5, 8, 32], "undo": 4, "unexpect": [6, 32], "unexpected": 25, "unexplain": 18, "unfair": [6, 29], "unfortun": [1, 8, 9, 10], "unicode_liter": [8, 9], "uniform": [0, 1, 5, 6, 11, 13, 23, 25, 28, 30, 31], "uniformli": [13, 25, 30, 31], "unifrompdf": 25, "unimport": [13, 30], "union": [5, 6, 32], "uniqu": [0, 2, 6, 13, 14, 22, 28, 32], "unique_cluster_label": 14, "unit": [0, 1, 3, 4, 5, 10, 12, 18, 25, 28, 29, 30, 31], "unitari": [5, 6, 22, 29, 30], "unitarili": [22, 28], "uniti": 25, "univari": 25, "univers": [0, 1, 2, 13, 21, 23, 24, 26, 28, 29, 30, 31, 32], "unix": 1, "unknow": [0, 22, 28], "unknown": [0, 1, 3, 4, 5, 6, 8, 10, 13, 22, 23, 28, 29, 30, 31, 32], "unknowwn": 12, "unlabel": 1, "unless": [0, 3, 6, 11, 13, 23, 28, 30, 32], "unlik": [1, 3, 8, 13, 30, 31], "unnecessarili": 9, "unord": 3, "unpickl": [], "unpleas": [], "unpublish": 31, "unravel": 1, "unrol": [3, 11], "unscal": 19, "unseen": [0, 7, 9, 15], "unstabl": 1, "unsupervis": [0, 1, 4, 12, 21, 28], "unsymmetr": [22, 28], "until": [1, 2, 4, 9, 12, 13, 14, 30, 31], "untouch": 0, "unusu": 12, "up": [1, 3, 4, 5, 6, 8, 10, 11, 13, 14, 16, 18, 19, 20, 21, 22, 23, 25, 26, 31], "updat": [1, 2, 10, 12, 13, 14, 15, 18, 32], "uploa": 28, "upload": [15, 20, 21, 23, 27], "upon": [0, 1, 6, 7, 11, 22], "upper": [0, 8, 9, 16, 22, 29], "uppercas": [22, 28], "upsampl": 4, "upscal": 4, "upward": [], "url": [28, 29], "us": [4, 5, 6, 8, 9, 10, 11, 12, 14, 15, 17, 20, 22, 25, 27, 32], "usag": [0, 8, 21, 28, 29], "usd": [], "usd10000": [], "use_bia": 4, "usecol": [0, 28], "useless": 1, "user": [0, 1, 2, 4, 6, 7, 15, 21, 22, 23, 28, 29], "usernam": [15, 23], "usetex": 25, "usg": 6, "usr": 25, "usual": [0, 3, 4, 7, 12, 13, 14, 28, 31], "ut": 5, "utf": [], "util": [1, 3, 4, 6, 7, 10, 14, 19, 28, 32], "ux": 22, "v": [2, 4, 5, 6, 11, 13, 15, 21, 29, 30, 32], "v0": 25, "v1": 25, "v2": 25, "v5": [], "v_": 31, "v_0": [11, 31], "v_t": 31, "va": 1, "vahid": 28, "val": 13, "val_accuraci": 3, "val_loss": 4, "vale": 2, "valid": [0, 1, 4, 7, 9, 10, 13, 21, 25, 28, 29, 31], "validation_data": 3, "validation_split": 4, "valu": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 12, 13, 14, 16, 17, 18, 20, 21, 22, 23, 28, 31], "valuat": 9, "valueerror": [], "valy": 4, "van": [0, 19, 23, 28, 29, 30, 31], "vandenbergh": [8, 13, 30], "vandermond": [0, 28], "vanilla": [0, 6, 11, 14, 29, 31], "vanish": [1, 4, 13, 25, 30], "var": [5, 6, 10, 11, 19, 23, 25, 29, 32], "var_x": 25, "varabl": 8, "varepsilon": [5, 6, 19, 32], "varepsilon_": [5, 6, 32], "varepsilon_i": [5, 6, 32], "vari": [0, 1, 3, 5, 6, 10, 28, 32], "variabl": [0, 1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 14, 22, 28, 29, 31, 32], "varianc": [0, 1, 5, 7, 9, 10, 11, 13, 14, 18, 20, 21, 22, 25, 28, 29, 30, 31], "variance_i": [5, 11, 29], "variance_x": [5, 11, 29], "variant": [0, 1, 6, 8, 12, 13, 28, 29, 30, 31], "variat": [3, 4, 11, 28], "varieti": [0, 3, 12, 21, 23, 28], "variou": [1, 3, 5, 6, 7, 8, 9, 11, 12, 13, 16, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31], "varydimens": 4, "vast": 31, "vastli": 3, "vaue": 1, "vault": 0, "vdot": [2, 13, 30, 31], "ve": [23, 31], "vec": [6, 32], "vector": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 13, 14, 17, 18, 21, 30, 31, 32], "vector_mean": 14, "ventur": [0, 8, 21, 28], "venv": 15, "verbos": [1, 3, 4], "veri": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 23, 25, 27, 28, 29, 30, 31, 32], "verifi": [3, 11, 22, 28], "versatil": [8, 28], "versicolor": [8, 9], "version": [0, 3, 10, 13, 14, 15, 21, 22, 23, 25, 28], "versu": [1, 31], "vert": [0, 1, 5, 6, 7, 8, 9, 11, 13, 16, 17, 28, 29, 30, 31, 32], "vert_1": [5, 6, 29, 30, 31], "vert_2": [5, 6, 11, 17, 29, 30, 31, 32], "via": [0, 5, 6, 7, 8, 9, 10, 11, 12, 19, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32], "vidal": 11, "video": [0, 1, 12, 21, 24, 26, 28, 29, 30], "view": [1, 3, 5, 6, 12, 13, 25, 27, 28, 30, 31, 32], "violat": 8, "virginica": 9, "viridi": [0, 1, 2, 3, 28], "virtanen": [], "virtual": [1, 31], "viscos": 13, "viscou": 13, "visibl": 15, "vision": [0, 3], "visit": 31, "visual": [0, 3, 11, 12, 18, 21, 28, 29], "visualis": 1, "visualstudio": [15, 16, 19], "viz": [6, 8, 25], "vmap": 13, "vmax": [1, 6], "vmh0zpt0tli": 31, "vmin": [1, 6], "voic": 3, "volatil": 31, "volum": [0, 3, 28], "vote": [10, 28], "voting_clf": 10, "votingclassifi": 10, "votingsimpl": 10, "vscode": [], "vstack": [5, 11, 22, 25, 28, 29], "vt": [5, 29, 30], "w": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 14, 22, 25, 28, 29, 30, 31, 32], "w1": 8, "w2": [8, 11], "w3": 8, "w_": [1, 12], "w_1": [8, 22], "w_1x_": 8, "w_1x_1": 8, "w_2": [8, 22], "w_2x_": 8, "w_2x_2": 8, "w_3": 22, "w_4": 22, "w_hidden": 2, "w_i": [1, 2, 10], "w_ix_i": 12, "w_j": 22, "w_m": 22, "w_output": 2, "w_px_": 8, "w_px_p": 8, "w_t": [], "wa": [1, 3, 4, 5, 6, 7, 10, 11, 12, 14, 17, 22, 28, 29, 31, 32], "wai": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 14, 15, 18, 19, 22, 25, 28, 29, 30, 31], "walk": 9, "walker": 25, "wall": 31, "walt": [], "wang": [0, 28], "want": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 20, 21, 23, 25, 28, 29, 30, 31, 32], "warn": 4, "warrant": [6, 32], "warranti": [], "wast": [3, 31], "watch": [21, 30, 31, 32], "wave": 3, "wavelet": 8, "wcag": [], "we": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27, 29, 30, 32], "weak": [9, 10, 14], "weaker": 31, "weather": [1, 12], "web": [21, 24, 26, 28], "webpag": 28, "websit": [6, 22, 23, 24, 28], "wedg": [8, 25], "wednesdai": [26, 28], "wee": 11, "week": [0, 5, 6, 7, 23, 24, 26], "weekli": [15, 16, 21, 23, 24, 26, 27, 28], "weight": [1, 2, 3, 6, 7, 9, 10, 12, 13, 18, 25, 31], "weigth": 2, "welcom": [8, 15, 21], "well": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 15, 16, 20, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "went": 8, "were": [0, 1, 3, 4, 5, 6, 7, 8, 10, 11, 12, 14, 25, 28, 31, 32], "wessel": [0, 19, 23, 28, 29, 30, 31], "what": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 19, 20, 21, 22, 23, 25, 31], "whatev": 3, "when": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 22, 23, 25, 28, 29, 30, 32], "whenev": [13, 15, 25, 31], "where": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 20, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "wherea": [6, 25, 31, 32], "wherefrom": 23, "wherein": [1, 12], "whether": [0, 3, 5, 7, 9, 23, 25, 28], "which": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 32], "whichev": [1, 3], "while": [0, 1, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 16, 19, 20, 25, 28, 29, 30, 31, 32], "white": 9, "whiteboad": 31, "whiteboard": [29, 30, 31], "who": [0, 15], "whole": [1, 3, 4, 5, 9, 11, 13, 31], "whom": [], "whose": [0, 6, 10, 25, 29, 32], "whow": [11, 29], "why": [0, 1, 3, 6, 13, 15, 16, 17, 19, 23, 29, 30], "wide": [0, 1, 3, 6, 7, 12, 21, 22, 23, 28, 32], "widehat": [6, 32], "width": [0, 3, 8, 9, 28], "wieringen": [0, 19, 23, 28, 29, 30, 31], "wiki": 23, "wikipedia": 23, "win": [10, 31], "wind": 9, "window": [], "wing": [26, 28], "winther": 2, "wiothout": 6, "wiscons": 7, "wisconsin": 10, "wisdom": [6, 29, 31], "wise": [1, 5, 12, 13, 29, 30, 31], "wish": [0, 2, 5, 7, 8, 11, 13, 14, 18, 22, 23, 28, 29, 30, 31], "with_std": [0, 29], "wither": 6, "within": [0, 2, 3, 4, 7, 9, 12, 13, 14, 25, 27, 28, 30], "withinclust": 14, "without": [0, 1, 5, 6, 8, 9, 11, 12, 13, 15, 18, 23, 28, 29, 30, 31, 32], "won": [0, 15, 28], "wonder": 8, "word": [0, 1, 3, 4, 5, 6, 7, 14, 19, 25, 28, 29, 30, 31], "work": [0, 1, 4, 6, 7, 8, 9, 13, 15, 16, 18, 19, 20, 21, 23, 24, 25, 26, 28, 29, 31, 32], "workabl": 31, "workaround": [], "workhors": 31, "workload": 31, "workshop": 28, "world": [0, 8, 16, 29], "worldwid": [0, 28], "worri": 15, "wors": [0, 1, 3, 4, 6, 28, 31, 32], "worth": [9, 19], "worthi": 23, "would": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 18, 20, 22, 23, 25, 28, 29, 30, 31, 32], "wouldn": [], "wrap": [6, 22, 28], "write": [0, 1, 2, 3, 5, 6, 7, 8, 12, 13, 15, 16, 18, 22, 28, 29, 31, 32], "written": [0, 2, 3, 5, 11, 12, 13, 16, 21, 22, 23, 25, 28, 29, 30, 31], "wrong": [1, 8, 15], "wrongli": 10, "wrote": [5, 11, 29], "wrt": [10, 13, 31], "wth": [10, 13, 31], "www": [20, 21, 22, 23, 27, 28, 30, 31, 32], "wx_1": 8, "x": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 28, 30, 31, 32], "x0": 8, "x1": [4, 8, 9, 10, 13], "x1_exampl": 8, "x1d": 8, "x2": [8, 9, 10, 13], "x2d": [8, 11], "x2d_train": 11, "x2dsl": 11, "x3": 8, "x_": [0, 2, 3, 5, 6, 8, 10, 11, 13, 14, 22, 25, 28, 29, 30, 31, 32], "x_0": [0, 5, 11, 18, 22, 28, 29, 32], "x_1": [0, 2, 5, 6, 7, 8, 9, 10, 11, 13, 18, 22, 25, 28, 29, 30, 31, 32], "x_2": [0, 2, 5, 6, 7, 8, 9, 10, 11, 13, 22, 25, 28, 29, 30, 32], "x_3": [8, 22, 25], "x_4": 22, "x_6": 18, "x_center": 11, "x_data": 1, "x_data_ful": 1, "x_hidden": 2, "x_i": [0, 1, 2, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 25, 28, 29, 30, 31, 32], "x_input": 2, "x_ix_": [0, 28], "x_iy_i": 8, "x_j": [0, 2, 8, 9, 12, 16, 25, 29, 31], "x_jy_j": 8, "x_k": [12, 14, 22, 25, 29], "x_l": 25, "x_m": [6, 12, 22, 25, 32], "x_mean": [18, 31], "x_n": [0, 2, 3, 6, 8, 11, 12, 13, 22, 25, 28, 30, 32], "x_new": [9, 10], "x_norm": [18, 31], "x_offset": [6, 29, 31], "x_output": 2, "x_p": [3, 7, 9], "x_poli": 9, "x_poly10": 9, "x_pred": 4, "x_prev": 2, "x_reduc": 11, "x_sampl": [], "x_scale": 8, "x_small": 13, "x_std": [18, 31], "x_t": 31, "x_test": [0, 1, 3, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 29, 30, 31, 32], "x_test_": 17, "x_test_own": 6, "x_test_scal": [0, 6, 7, 9, 10, 11, 29, 31], "x_tot": 4, "x_train": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "x_train_": 17, "x_train_mean": [6, 29, 31], "x_train_own": 6, "x_train_r": 19, "x_train_scal": [0, 6, 7, 9, 10, 11, 29, 31], "x_val": 1, "xarrai": [21, 28], "xavier": 1, "xbnew": [13, 30, 31], "xcode": [0, 21, 23, 28], "xdclassiffierconfus": 10, "xdclassiffierroc": 10, "xg_clf": 10, "xgb": 10, "xgbclassifi": 10, "xgboost": 9, "xgboot": 10, "xgbregressor": 10, "xgparam": 10, "xgtree": 10, "xi": [8, 13, 31], "xi_": 8, "xi_1": 8, "xi_i": 8, "xk": 8, "xla": [13, 21, 28], "xlabel": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 25, 28, 29, 30, 31, 32], "xlim": [6, 10, 32], "xm": 9, "xmesh": 13, "xnew": [0, 13, 28, 30, 31], "xp": 25, "xpanda": [0, 29], "xpd": [5, 11, 29], "xplot": 0, "xscale": [0, 29], "xsr": 9, "xt_x": [13, 30, 31], "xtest": [6, 32], "xtick": [3, 6, 8, 9, 32], "xtrain": [6, 32], "xu": [0, 28], "xx": [0, 22, 28], "xy": [0, 6, 8, 22, 28], "xytext": 8, "xyz": [], "xz": [22, 28], "y": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "y1": 4, "y2": 4, "y3": 4, "y_": [0, 1, 5, 6, 10, 11, 22, 28, 29, 32], "y_0": [0, 5, 11, 22, 28, 29, 32], "y_1": [0, 5, 8, 9, 11, 13, 22, 28, 29, 30, 31, 32], "y_1y_1": 8, "y_1y_1k": 8, "y_1y_2": 8, "y_1y_2k": 8, "y_1y_n": 8, "y_1y_nk": 8, "y_2": [0, 5, 8, 9, 11, 22, 28, 29], "y_2y_1": 8, "y_2y_1k": 8, "y_2y_2": 8, "y_2y_2k": 8, "y_3": [0, 9, 22], "y_4": 22, "y_center": [18, 31], "y_data": [0, 1, 5, 6, 28, 29, 30, 31], "y_data_ful": 1, "y_decis": 8, "y_fit": [0, 29], "y_i": [0, 1, 5, 6, 7, 8, 9, 10, 11, 12, 13, 19, 22, 23, 28, 29, 30, 31, 32], "y_if_": 10, "y_ix_": [0, 28], "y_ix_i": [7, 8, 13, 29, 30, 31], "y_iy_jk": 8, "y_j": [6, 8, 12, 23, 32], "y_k": 12, "y_m": 22, "y_mean": [18, 31], "y_model": [0, 4, 5, 6, 28, 29, 30, 31], "y_n": [8, 13, 30, 31], "y_ny_1": 8, "y_ny_1k": 8, "y_ny_2": 8, "y_ny_2k": 8, "y_ny_n": 8, "y_ny_nk": 8, "y_offset": [6, 17, 29, 31], "y_plot": 9, "y_pred": [0, 1, 4, 6, 7, 8, 9, 10, 29, 31, 32], "y_pred1": 9, "y_pred2": 9, "y_pred_rf": 10, "y_pred_tre": 10, "y_proba": [7, 10], "y_sampl": [], "y_scaler": [6, 29, 31], "y_test": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 29, 30, 31, 32], "y_test_onehot": 1, "y_test_predict": [], "y_tot": 4, "y_train": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "y_train_mean": [6, 29, 31], "y_train_onehot": 1, "y_train_predict": [], "y_train_r": 19, "y_train_scal": [6, 29, 31], "y_val": 1, "ye": [3, 6, 7, 32], "year": [0, 21, 28], "yet": [0, 1, 6, 8, 11, 13, 20, 28], "yi": [13, 31], "yield": [0, 2, 5, 6, 8, 10, 12, 13, 14, 22, 25, 28, 30, 31, 32], "yk": 8, "ylabel": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 25, 28, 29, 30, 31, 32], "ylim": [3, 6, 32], "ym": 9, "ymesh": 13, "yn": 0, "yo": [8, 9, 10], "yoshiki": [], "yoshua": [1, 27], "you": [0, 1, 3, 4, 5, 6, 8, 9, 10, 11, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27, 28, 29, 30, 31, 32], "young": 0, "your": [1, 2, 4, 5, 6, 8, 11, 13, 15, 17, 19, 20, 21, 22, 28, 30, 31, 32], "your_model_object": 16, "yourself": [11, 13, 28, 30], "youtu": [29, 30, 32], "youtub": [21, 30, 31, 32], "ypred": [6, 32], "ypredict": [0, 13, 28, 29, 30, 31], "ypredict2": [13, 30, 31], "ypredictlasso": [5, 30], "ypredictol": [0, 5, 30], "ypredictown": [6, 29, 31], "ypredictownridg": [6, 29, 30, 31], "ypredictridg": [0, 5, 6, 29, 30, 31], "ypredictskl": [6, 29, 31], "ytest": [6, 32], "ytick": [3, 6, 8, 9, 32], "ytild": [0, 6, 28, 29, 32], "ytildelasso": [5, 30], "ytildenp": [0, 28, 29], "ytildeol": [0, 5, 30], "ytildeownridg": [6, 29, 30, 31], "ytilderidg": [5, 6, 29, 30, 31], "ytrain": [6, 32], "yuxi": 28, "yx": [22, 28], "yy": [22, 28], "yz": [22, 28], "z": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 22, 25, 28, 29, 32], "z_": [1, 2, 12, 22, 28], "z_0": [22, 28], "z_1": [22, 28], "z_2": [22, 28], "z_c": 1, "z_h": 1, "z_hidden": 2, "z_i": [1, 12], "z_j": [1, 12], "z_k": [12, 29], "z_m": 1, "z_mod": 9, "z_o": 1, "z_output": 2, "za": [], "zaman": 25, "zaxi": 6, "zero": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "zeros_lik": 4, "zeroth": 29, "zfill": 4, "zip": [4, 6], "zm_h": [0, 28], "zn": [], "zone": [], "zoom": 28, "zscout": [], "zx": [22, 28], "zy": [22, 28], "zz": [22, 28], "\u00f8yvind": [6, 29, 31]}, "titles": ["3. Linear Regression", "14. Building a Feed Forward Neural Network", "15. Solving Differential Equations with Deep Learning", "16. Convolutional Neural Networks", "17. Recurrent neural networks: Overarching view", "4. Ridge and Lasso Regression", "5. Resampling Methods", "6. Logistic Regression", "8. Support Vector Machines, overarching aims", "9. Decision trees, overarching aims", "10. Ensemble Methods: From a Single Tree to Many Trees and Extreme Boosting, Meet the Jungle of Methods", "11. Basic ideas of the Principal Component Analysis (PCA)", "13. Neural networks", "7. Optimization, the central part of any Machine Learning algortithm", "12. Clustering and Unsupervised Learning", "Exercises week 34", "Exercises week 35", "Exercises week 36", "Exercises week 37", "Exercises week 38", "Exercises week 39", "Applied Data Analysis and Machine Learning", "2. Linear Algebra, Handling of Arrays and more Python Features", "Project 1 on Machine Learning, deadline October 6 (midnight), 2025", "Course setting", "1. Elements of Probability Theory and Statistical Data Analysis", "Teachers and Grading", "Textbooks", "Week 34: Introduction to the course, Logistics and Practicalities", "Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression", "Week 36: Linear Regression and Gradient descent", "Week 37: Gradient descent methods", "Week 38: Statistical analysis, bias-variance tradeoff and resampling methods"], "titleterms": {"": [8, 10, 30, 31, 32], "0": [], "04": [], "05": [], "06": [], "07": [], "1": [0, 15, 16, 17, 18, 19, 20, 23, 29], "11": [], "15": [19, 32], "19": 19, "1a": 18, "2": [0, 15, 16, 17, 18, 19, 20, 28, 29, 30], "20": [], "2017": [], "2018": [], "2019": [], "2023": 26, "2025": 23, "27": [], "2a": [], "2b": [], "3": [0, 15, 16, 17, 18, 19, 20, 29], "34": [15, 28], "35": [16, 29], "36": [17, 30], "37": [18, 31], "38": [19, 32], "39": 20, "3a": 18, "3b": 18, "4": [0, 15, 16, 17, 18, 19, 20, 29], "4a": 18, "4b": 18, "5": [0, 16, 18, 19, 20], "6": 23, "8": 31, "A": [0, 1, 4, 8, 9, 28, 32], "And": [28, 29, 31], "But": 31, "In": 26, "Ising": 6, "The": [0, 1, 2, 3, 5, 6, 7, 8, 9, 11, 12, 15, 21, 28, 29, 30, 31, 32], "To": 28, "With": [4, 30], "a11i": [], "about": [28, 29, 30], "abov": 30, "abstract": 20, "accuraci": 31, "across": 31, "activ": [1, 12], "ad": [0, 6, 20, 23, 28, 29], "adaboost": 10, "adagrad": [13, 31], "adam": [13, 31], "adapt": [10, 31], "add": [], "adjust": 1, "advanc": 23, "adversari": 4, "again": [3, 9], "ai": [23, 28], "aim": [8, 9, 28], "aka": 28, "al": 31, "algebra": [22, 28], "algorithm": [9, 10, 11, 12, 28, 29, 30, 31], "algortithm": [13, 30], "all": 8, "an": [0, 4, 10, 15, 20, 28], "analys": [5, 29, 30], "analysi": [0, 5, 6, 11, 21, 23, 25, 28, 29, 30, 32], "analyt": [0, 16, 18], "ani": [13, 30], "anoth": [9, 30, 32], "api": [], "appli": 21, "approach": [0, 8, 14, 28, 31, 32], "approxim": 12, "architectur": 1, "arrai": [22, 28], "assist": 26, "assumpt": 32, "august": [], "author": [], "autocorrel": 25, "autograd": [2, 13, 31], "automat": [13, 31], "avail": 20, "avali": [], "averag": 31, "b": 23, "back": [1, 11, 12, 29, 30], "background": [21, 23, 32], "bag": 10, "base": [13, 31, 32], "basic": [0, 5, 7, 9, 10, 11, 22, 29, 30], "batch": [1, 31], "bay": 5, "befor": 11, "beta": 32, "better": 8, "bia": [6, 19, 23, 31, 32], "binari": 1, "bind": 28, "bird": 10, "blind": [], "block": [], "boldsymbol": [18, 29, 32], "book": 19, "boost": 10, "bootstrap": [6, 10, 32], "boston": [], "breast": 1, "brief": [28, 32], "bring": 12, "browser": [], "bsd": [], "build": [1, 3, 9], "c": [23, 28], "calcul": [18, 29, 30], "can": [28, 31, 32], "cancer": [1, 7, 9, 11], "cart": 9, "case": [8, 10, 25, 29, 30, 31], "cdn": [], "cell": [], "central": [13, 21, 25, 30, 32], "chain": 12, "challeng": 31, "chang": 10, "changelog": [], "channel": 28, "chi": [0, 28], "choic": 17, "choos": [1, 31], "cifar01": 3, "citat": [], "classic": 11, "classif": [1, 9, 10], "classifi": 8, "claus": [], "clip": 1, "cluster": 14, "cnn": 3, "code": [1, 2, 5, 9, 11, 12, 13, 14, 15, 16, 20, 23, 28, 29, 30, 31, 32], "collect": [1, 3], "color": [], "colorblind": [], "combin": 31, "commun": 28, "compar": [2, 10, 16], "comparison": [30, 31], "compet": 31, "compil": [], "complet": 29, "complex": [0, 6, 23, 29], "complic": 6, "compon": 11, "comput": [9, 19, 31], "computation": 32, "computerlab": 28, "con": [9, 31], "concept": 25, "condit": 30, "confid": 32, "conjug": 13, "constraint": 31, "contain": [], "content": [], "contn": 28, "contrast": [], "contributor": [], "converg": 31, "convex": [8, 13, 30, 31], "convolut": [3, 12], "copyright": [], "core": [], "correct": 31, "correl": [11, 29], "correspond": [], "cost": [1, 10, 29, 30, 31, 32], "cours": [21, 24, 27, 28], "covari": [5, 11, 25, 29], "cover": 28, "creat": [16, 20], "creator": [], "cross": [6, 23, 32], "cython": 28, "d": 23, "dark": [], "data": [0, 1, 3, 6, 7, 9, 11, 15, 17, 18, 21, 25, 28, 29], "dataset": [1, 3, 18], "david": 28, "deadlin": [23, 28], "deadllin": 26, "decai": [2, 31], "decis": [9, 10], "decomposit": [5, 11, 22, 29, 30], "deeep": [], "deep": [1, 2, 28, 31], "defin": [1, 28], "definit": 19, "deflist": [], "degre": [0, 17, 29], "deliver": [15, 16, 19, 20], "deliveri": 23, "delta": 32, "dens": 0, "depend": [], "deriv": [5, 12, 16, 17, 19, 29, 30, 31, 32], "descent": [2, 10, 13, 18, 23, 30, 31], "design": 29, "detail": [3, 28], "develop": 1, "diagon": 11, "differ": [8, 31], "differenti": [2, 13, 31], "diffus": 2, "dimens": 31, "dimension": [2, 3, 8, 18], "direct": [], "disadvantag": 9, "discret": 25, "discrimin": 28, "distribut": [5, 25, 32], "do": [1, 31], "document": 20, "doe": [29, 30], "domain": 25, "down": 1, "dropout": 1, "e": 23, "economi": [29, 30], "electron": 23, "element": [0, 25, 28], "elimin": 22, "empir": 31, "energi": 28, "ensembl": 10, "entropi": 9, "environ": [0, 15], "equat": [0, 2, 12, 29, 30], "error": [0, 10, 28, 29, 30, 32], "essenti": 28, "estim": 32, "et": 31, "etc": 28, "euler": 2, "evalu": 1, "evid": 31, "exampl": [1, 2, 3, 4, 6, 7, 8, 9, 10, 28, 29, 30, 31, 32], "exercis": [0, 6, 15, 16, 17, 18, 19, 20, 29], "expect": [19, 25, 32], "expens": 32, "experi": 25, "explor": 0, "exponenti": [2, 31], "express": [16, 17, 19, 29], "extend": 30, "extrapol": 4, "extrem": [10, 28], "ey": 10, "f": 23, "fall": 26, "famili": [1, 28], "famou": 22, "fantast": [29, 30], "faq": [], "featur": [9, 16, 22, 29], "februari": [], "feed": [1, 12], "figur": 20, "file": [], "fill": [], "final": [12, 29, 31], "find": [16, 18, 32], "fine": 1, "first": [4, 12, 28, 30], "fit": [0, 10, 15, 16, 28, 30], "fix": [29, 30, 31], "fold": 32, "forc": 3, "forest": 10, "form": 18, "format": [23, 28], "formula": 18, "forward": [1, 2, 12], "foster": 28, "fourier": 3, "frank": 6, "freedom": [0, 17, 29], "frequent": [29, 31], "frequentist": [0, 28], "from": [5, 10, 12, 28, 29, 30, 31, 32], "full": [2, 31], "function": [0, 1, 6, 7, 8, 10, 11, 12, 13, 23, 25, 28, 29, 30, 31, 32], "further": [3, 5, 29, 30], "g": 23, "gan": 4, "gaussian": 22, "gd": [13, 31], "gener": [4, 9, 28], "geometr": [11, 30], "get": 20, "gini": 9, "github": 15, "goal": [15, 16, 17, 18, 19, 20], "good": [0, 20, 28], "goodfellow": 31, "gotthard": [], "grade": [26, 28], "gradient": [1, 2, 10, 13, 18, 23, 30, 31], "greativ": [], "growth": 2, "guid": [], "h": 23, "ha": 21, "handl": [22, 28], "happen": 32, "hessian": [29, 30, 31], "hidden": 2, "high": [], "histogram": 32, "histori": [], "hous": [], "how": 16, "hyperparamet": [1, 17], "hyperplan": 8, "i": [0, 1, 28], "id3": 9, "idea": 11, "ideal": 30, "ident": 32, "identifi": 32, "ii": 28, "iid": 32, "illustr": 30, "implement": [1, 16, 17, 18], "implic": [5, 29, 30], "import": [5, 22, 28, 29, 30], "improv": [1, 31], "includ": [13, 23, 31], "incorpor": [], "increment": 11, "independ": 32, "index": 9, "inform": 26, "input": 2, "instal": [21, 23, 28], "instructor": 26, "interpret": [5, 11, 19, 28, 29, 30, 32], "interv": 32, "introduc": [11, 13, 29], "introduct": [0, 6, 20, 21, 22, 23, 28], "invers": [5, 22], "invert": [29, 30], "ipython": [], "iter": 10, "its": 29, "j": [], "jacobian": 29, "januari": [], "jax": 13, "julia": 28, "jungl": 10, "jupyt": [], "k": 32, "kera": [1, 3], "kernel": [8, 11], "lab": [30, 31, 32], "lagrangian": 8, "lasso": [5, 6, 23, 29, 30], "last": [29, 31], "later": [5, 29, 30], "layer": [1, 2, 3, 12], "learn": [0, 1, 2, 11, 13, 14, 15, 16, 17, 18, 19, 20, 21, 23, 28, 29, 30, 31, 32], "least": [5, 6, 16, 19, 23, 28, 29, 30, 31], "lectur": [28, 30, 31, 32], "level": 10, "librari": [21, 28], "licens": [], "light": [], "likelihood": [7, 32], "limit": [1, 13, 25, 30, 31, 32], "linear": [0, 8, 13, 15, 22, 28, 29, 30], "link": [5, 11, 27, 29, 32], "literatur": 23, "logist": [7, 28], "loss": [29, 30, 31], "lu": 22, "ma": [], "machin": [0, 8, 13, 21, 23, 28, 30], "made": 32, "main": [25, 28], "make": [0, 9, 10, 20, 29], "mani": [10, 12], "markdown": [], "mask": [], "maskedarrai": [], "mass": 28, "materi": [23, 28, 29, 30, 31, 32], "math": [5, 29, 30], "mathemat": [3, 5, 8, 29, 30], "matplotlib": [], "matric": [5, 22, 28], "matrix": [1, 5, 11, 12, 16, 22, 28, 29, 30, 31], "matter": 0, "max": 29, "maximum": 32, "me": [], "mean": [0, 29, 30], "meet": [5, 10, 25, 28, 29], "memori": 31, "mercer": 8, "metadata": [], "method": [6, 9, 10, 13, 23, 28, 30, 31, 32], "metric": 19, "midnight": 23, "min": 29, "mini": 31, "minibatch": 31, "minim": 28, "mit": [], "ml": 28, "mle": 32, "mlp": 12, "mnist": [3, 4], "model": [0, 1, 4, 6, 12, 15, 17, 28], "moment": 31, "momentum": [13, 23, 31], "mondai": [30, 31, 32], "moon": [8, 9], "more": [3, 6, 22, 23, 28, 29, 30, 31, 32], "motiv": 31, "move": 31, "multilay": 12, "multipl": [1, 3, 17], "multipli": 8, "myst": [], "ncsa": [], "need": [23, 28], "network": [1, 2, 3, 4, 7, 12, 28, 31], "neural": [1, 2, 3, 4, 7, 12, 28, 31], "new": [4, 18, 32], "newton": [30, 31], "node": [], "non": [8, 31], "none": 31, "normal": [0, 1, 32], "notat": 12, "note": [23, 29, 30], "notebook": [], "novemb": [], "now": [1, 9, 13, 30, 31, 32], "nuclear": [0, 28], "numba": 28, "number": [0, 2, 25, 29, 31], "numer": [2, 23, 25], "numpi": [22, 28], "object": 3, "obtain": 11, "octob": 23, "od": 2, "off": [6, 19, 23], "ol": [5, 6, 15, 16, 18, 23, 30, 32], "one": [2, 12, 18, 30], "open": [], "oper": 22, "optim": [1, 8, 13, 18, 21, 28, 29, 30, 31], "order": [13, 18, 31], "ordinari": [5, 6, 16, 19, 23, 28, 29, 30, 31], "organ": [0, 28], "oslo": 27, "other": [4, 9, 11, 12, 22, 23, 28], "our": [0, 4, 5, 11, 13, 23, 28, 29, 30], "outcom": [21, 28], "output": 2, "overarch": [0, 4, 8, 9, 28, 29], "overview": [10, 28, 31], "own": [0, 10, 11, 23, 28, 29], "packag": [22, 28], "panda": [28, 29], "paramet": [28, 29], "paramt": 18, "part": [13, 21, 23, 30], "partial": 2, "pass": 1, "pca": 11, "pdf": 25, "perceptron": 12, "perform": [1, 9], "period": 3, "perspect": 1, "pitaya": [], "plan": [29, 30, 31, 32], "plethora": 28, "plot": 32, "point": 4, "poisson": 2, "polici": [], "polynomi": [3, 16, 18, 30], "popul": 2, "popular": 28, "practic": [13, 26, 28, 31], "pre": [1, 3], "preambl": 23, "predict": 4, "preprocess": [29, 31], "prerequisit": [3, 21, 28], "present": 20, "princip": 11, "principl": 3, "pro": [9, 31], "probabl": [5, 25, 32], "problem": [1, 2, 13, 28, 29, 30, 31], "procedur": [9, 28], "process": [1, 3], "program": [2, 13, 23, 30, 31], "project": [6, 20, 23, 26, 28], "prop": 13, "propag": [1, 12], "properti": [5, 25, 29, 30, 31], "python": [0, 9, 15, 21, 22, 28], "quick": 8, "quickli": [], "r": 28, "random": [10, 11, 25], "raphson": 30, "rate": [23, 31], "read": [9, 28, 29, 31, 32], "real": [6, 28], "recommend": [28, 29], "record": [], "recurr": [4, 12], "reduc": [0, 29], "reduct": 3, "refer": 23, "referenc": 20, "reformul": 2, "regress": [0, 5, 6, 7, 9, 10, 13, 15, 17, 18, 19, 23, 28, 29, 30, 31, 32], "regular": 1, "relat": [], "relev": [27, 29], "relu": 1, "remark": 3, "remind": [6, 8, 28, 29, 30, 31], "replac": [13, 31], "report": [20, 23], "repositori": [15, 32], "requir": [2, 21], "resampl": [6, 19, 23, 32], "rescal": [6, 29], "residu": [29, 30], "resourc": 2, "result": [29, 30], "revis": [], "revisit": [13, 30, 31], "rewrit": [28, 29, 32], "ridg": [0, 5, 6, 17, 18, 19, 23, 29, 30, 31], "rm": 13, "rmsprop": 31, "role": [], "rule": [12, 31], "rung": 23, "same": [13, 31, 32], "sampl": 11, "scalabl": 31, "scale": [17, 18, 19, 29, 31], "schedul": 28, "schemat": 9, "scheme": 2, "scienc": 28, "scikit": [0, 1, 11, 28, 29, 30, 31, 32], "second": [13, 18, 31], "semest": 26, "sensit": 30, "septemb": [19, 30, 31, 32], "session": [30, 31, 32], "set": [0, 2, 3, 9, 12, 15, 24, 28, 29, 30], "setup": 15, "sgd": [13, 31], "should": 1, "show": [], "similar": [13, 31], "simpl": [0, 4, 9, 13, 18, 28, 29, 30, 31], "simplest": 18, "singl": 10, "singular": [5, 11, 29, 30], "size": [29, 30, 31], "sklearn": 16, "slightli": 31, "smoothi": [], "sneak": 31, "soft": 8, "softmax": 1, "softwar": [23, 28], "solv": [2, 30], "solver": 13, "some": [13, 22, 29, 30], "sourc": [], "specifi": 2, "speed": 31, "sphinx": [], "split": [0, 15, 29], "squar": [0, 5, 6, 10, 16, 19, 23, 28, 29, 30, 31], "standard": [13, 29, 32], "start": 20, "state": 0, "statist": [5, 6, 21, 25, 28, 32], "steepest": [10, 13, 30], "step": [31, 32], "stochast": [13, 23, 25, 31], "stop": 31, "strongli": [28, 31], "structur": [], "suggest": 28, "sum": 32, "summari": [26, 28], "superposit": 3, "supervis": 1, "support": 8, "svd": [5, 29, 30], "synthet": 18, "systemat": 3, "t": 29, "take": 16, "taken": [28, 31], "teach": 26, "teacher": [26, 28], "team": [], "technic": 29, "techniqu": [6, 11, 23], "technologi": 21, "tensorflow": [1, 3], "tent": [26, 28], "term": 32, "test": [0, 1, 15, 17, 29], "texmath": [], "text": 28, "textbook": [27, 28], "than": 30, "thank": [], "theorem": [5, 8, 11, 12, 25, 32], "theoret": 31, "theori": 25, "theta": 18, "thi": 28, "time": 31, "tip": [13, 31], "todo": [], "togeth": 12, "tool": [23, 28], "top": 1, "topic": 28, "toward": 11, "trade": [6, 19, 23], "tradeoff": [6, 32], "train": [0, 1, 4, 15, 28, 29], "transform": 3, "translat": [], "tree": [9, 10], "tuesdai": 30, "tune": 1, "two": [3, 8, 21], "type": [2, 4, 12, 28], "uio": 28, "understand": 32, "univers": [12, 27], "unsupervis": 14, "up": [0, 2, 9, 12, 15, 28, 29, 30, 32], "updat": [23, 31], "us": [0, 1, 2, 3, 7, 13, 16, 18, 19, 21, 23, 28, 29, 30, 31], "usag": 31, "v": [3, 31], "valid": [6, 23, 32], "valu": [5, 11, 19, 25, 29, 30, 32], "vari": 31, "variabl": [25, 30], "varianc": [6, 19, 23, 32], "variou": [0, 32], "vector": [8, 12, 16, 22, 28, 29], "versu": 28, "video": [31, 32], "view": [0, 4, 10, 29], "virtual": 15, "visual": [1, 9], "wai": [9, 23, 32], "wave": 2, "we": [28, 31], "wednesdai": 30, "week": [15, 16, 17, 18, 19, 20, 28, 29, 30, 31, 32], "weekli": [], "welcom": [], "what": [0, 28, 29, 30, 32], "when": 31, "which": [1, 31], "why": [28, 31, 32], "wisconsin": 7, "workflow": [], "wrap": 32, "write": [4, 11, 20, 23, 30], "x": 29, "xgboost": 10, "yaml": [], "yet": 30, "your": [0, 10, 16, 18, 23, 29]}}) \ No newline at end of file +Search.setIndex({"alltitles": {"1a)": [[18, "a"]], "3a)": [[18, "id1"]], "3b)": [[18, "b"]], "4a)": [[18, "id2"]], "4b)": [[18, "id3"]], "A Classification Tree": [[9, "a-classification-tree"]], "A Frequentist approach to data analysis": [[0, "a-frequentist-approach-to-data-analysis"], [28, "a-frequentist-approach-to-data-analysis"]], "A better approach": [[8, "a-better-approach"]], "A first summary": [[28, "a-first-summary"]], "A new Cost Function": [[32, "a-new-cost-function"]], "A quick Reminder on Lagrangian Multipliers": [[8, "a-quick-reminder-on-lagrangian-multipliers"]], "A simple example": [[4, "a-simple-example"]], "A soft classifier": [[8, "a-soft-classifier"]], "A top-down perspective on Neural networks": [[1, "a-top-down-perspective-on-neural-networks"]], "A way to Read the Bias-Variance Tradeoff": [[32, "a-way-to-read-the-bias-variance-tradeoff"]], "ADAM algorithm, taken from Goodfellow et al": [[31, "adam-algorithm-taken-from-goodfellow-et-al"]], "ADAM optimizer": [[13, "adam-optimizer"], [31, "id2"]], "Accuracy": [[31, "accuracy"]], "Activation functions": [[12, "activation-functions"]], "AdaGrad Properties": [[31, "adagrad-properties"]], "AdaGrad Update Rule Derivation": [[31, "adagrad-update-rule-derivation"]], "AdaGrad algorithm, taken from Goodfellow et al": [[31, "adagrad-algorithm-taken-from-goodfellow-et-al"]], "Adam Optimizer": [[31, "adam-optimizer"]], "Adam vs. AdaGrad and RMSProp": [[31, "adam-vs-adagrad-and-rmsprop"]], "Adam: Bias Correction": [[31, "adam-bias-correction"]], "Adam: Exponential Moving Averages (Moments)": [[31, "adam-exponential-moving-averages-moments"]], "Adam: Update Rule Derivation": [[31, "adam-update-rule-derivation"]], "Adaptive boosting: AdaBoost, Basic Algorithm": [[10, "adaptive-boosting-adaboost-basic-algorithm"]], "Adaptivity Across Dimensions": [[31, "adaptivity-across-dimensions"]], "Adding error analysis and training set up": [[28, "adding-error-analysis-and-training-set-up"], [29, "adding-error-analysis-and-training-set-up"]], "Adjust hyperparameters": [[1, "adjust-hyperparameters"]], "Algorithms and codes for Adagrad, RMSprop and Adam": [[31, "algorithms-and-codes-for-adagrad-rmsprop-and-adam"]], "Algorithms for Setting up Decision Trees": [[9, "algorithms-for-setting-up-decision-trees"]], "An Overview of Ensemble Methods": [[10, "an-overview-of-ensemble-methods"]], "An extrapolation example": [[4, "an-extrapolation-example"]], "An optimization/minimization problem": [[28, "an-optimization-minimization-problem"]], "And finally \\boldsymbol{X}\\boldsymbol{X}^T": [[29, "and-finally-boldsymbol-x-boldsymbol-x-t"]], "And finally ADAM": [[31, "and-finally-adam"]], "And what about using neural networks?": [[28, "and-what-about-using-neural-networks"]], "Another Example from Scikit-Learn\u2019s Repository": [[32, "another-example-from-scikit-learn-s-repository"]], "Another Example, now with a polynomial fit": [[30, "another-example-now-with-a-polynomial-fit"]], "Another example, the moons again": [[9, "another-example-the-moons-again"]], "Applied Data Analysis and Machine Learning": [[21, null]], "Assumptions made": [[32, "assumptions-made"]], "Autocorrelation function": [[25, "autocorrelation-function"]], "Automatic differentiation": [[13, "automatic-differentiation"]], "Back to Ridge and LASSO Regression": [[29, "back-to-ridge-and-lasso-regression"], [30, "back-to-ridge-and-lasso-regression"]], "Back to the Cancer Data": [[11, "back-to-the-cancer-data"]], "Background literature": [[23, "background-literature"]], "Bagging": [[10, "bagging"]], "Bagging Examples": [[10, "bagging-examples"]], "Basic Matrix Features": [[22, "basic-matrix-features"]], "Basic ideas of the Principal Component Analysis (PCA)": [[11, null]], "Basic math of the SVD": [[5, "basic-math-of-the-svd"], [29, "basic-math-of-the-svd"], [30, "basic-math-of-the-svd"]], "Basics": [[7, "basics"]], "Basics of a tree": [[9, "basics-of-a-tree"]], "Batch Normalization": [[1, "batch-normalization"]], "Batches and mini-batches": [[31, "batches-and-mini-batches"]], "Bayes\u2019 Theorem and Ridge and Lasso Regression": [[5, "bayes-theorem-and-ridge-and-lasso-regression"]], "Boosting, a Bird\u2019s Eye View": [[10, "boosting-a-bird-s-eye-view"]], "Bootstrap": [[6, "bootstrap"]], "Bringing it together, first back propagation equation": [[12, "bringing-it-together-first-back-propagation-equation"]], "Building a Feed Forward Neural Network": [[1, null]], "Building a tree, regression": [[9, "building-a-tree-regression"]], "Building neural networks in Tensorflow and Keras": [[1, "building-neural-networks-in-tensorflow-and-keras"]], "But none of these can compete with Newton\u2019s method": [[31, "but-none-of-these-can-compete-with-newton-s-method"]], "CNNs in more detail, building convolutional neural networks in Tensorflow and Keras": [[3, "cnns-in-more-detail-building-convolutional-neural-networks-in-tensorflow-and-keras"]], "Cancer Data again now with Decision Trees and other Methods": [[9, "cancer-data-again-now-with-decision-trees-and-other-methods"]], "Challenge: Choosing a Fixed Learning Rate": [[31, "challenge-choosing-a-fixed-learning-rate"]], "Choose cost function and optimizer": [[1, "choose-cost-function-and-optimizer"]], "Classical PCA Theorem": [[11, "classical-pca-theorem"]], "Clustering and Unsupervised Learning": [[14, null]], "Code Example for Cross-validation and k-fold Cross-validation": [[32, "code-example-for-cross-validation-and-k-fold-cross-validation"]], "Code example for the Bootstrap method": [[32, "code-example-for-the-bootstrap-method"]], "Code for SVD and Inversion of Matrices": [[5, "code-for-svd-and-inversion-of-matrices"]], "Code with a Number of Minibatches which varies": [[31, "code-with-a-number-of-minibatches-which-varies"]], "Codes and Approaches": [[14, "codes-and-approaches"]], "Codes for the SVD": [[5, "codes-for-the-svd"], [29, "codes-for-the-svd"], [30, "codes-for-the-svd"]], "Coding Setup and Linear Regression": [[15, "coding-setup-and-linear-regression"]], "Collect and pre-process data": [[1, "collect-and-pre-process-data"]], "Communication channels": [[28, "communication-channels"]], "Compare Bagging on Trees with Random Forests": [[10, "compare-bagging-on-trees-with-random-forests"]], "Comparing with a numerical scheme": [[2, "comparing-with-a-numerical-scheme"]], "Comparison with OLS": [[30, "comparison-with-ols"]], "Computation of gradients": [[31, "computation-of-gradients"]], "Computing the Gini index": [[9, "computing-the-gini-index"]], "Conditions on convex functions": [[30, "conditions-on-convex-functions"]], "Confidence Intervals": [[32, "confidence-intervals"]], "Conjugate gradient method": [[13, "conjugate-gradient-method"]], "Convergence rates": [[31, "convergence-rates"]], "Convex function": [[30, "convex-function"]], "Convex functions": [[13, "convex-functions"], [30, "convex-functions"]], "Convolution Examples: Polynomial multiplication": [[3, "convolution-examples-polynomial-multiplication"]], "Convolution Examples: Principle of Superposition and Periodic Forces (Fourier Transforms)": [[3, "convolution-examples-principle-of-superposition-and-periodic-forces-fourier-transforms"]], "Convolutional Neural Network": [[12, "convolutional-neural-network"]], "Convolutional Neural Networks": [[3, null]], "Correlation Function and Design/Feature Matrix": [[29, "correlation-function-and-design-feature-matrix"]], "Correlation Matrix": [[11, "correlation-matrix"], [29, "correlation-matrix"]], "Correlation Matrix with Pandas": [[29, "correlation-matrix-with-pandas"]], "Course Format": [[28, "course-format"]], "Course setting": [[24, null]], "Covariance Matrix Examples": [[29, "covariance-matrix-examples"]], "Covariance and Correlation Matrix": [[29, "covariance-and-correlation-matrix"]], "Cross-validation": [[6, "cross-validation"]], "Cross-validation in brief": [[32, "cross-validation-in-brief"]], "Deadlines for projects (tentative)": [[28, "deadlines-for-projects-tentative"]], "Decision trees, overarching aims": [[9, null]], "Deep Neural Networks": [[31, "deep-neural-networks"]], "Deep learning methods": [[28, "deep-learning-methods"]], "Define model and architecture": [[1, "define-model-and-architecture"]], "Defining the cost function": [[1, "defining-the-cost-function"]], "Definitions": [[19, "definitions"]], "Deliverables": [[15, "deliverables"], [16, "deliverables"], [19, "deliverables"], [20, "deliverables"]], "Derivation of the AdaGrad Algorithm": [[31, "derivation-of-the-adagrad-algorithm"]], "Derivatives and the chain rule": [[12, "derivatives-and-the-chain-rule"]], "Derivatives, example 1": [[29, "derivatives-example-1"]], "Deriving OLS from a probability distribution": [[5, "deriving-ols-from-a-probability-distribution"], [32, "deriving-ols-from-a-probability-distribution"]], "Deriving and Implementing Ordinary Least Squares": [[16, "deriving-and-implementing-ordinary-least-squares"]], "Deriving and Implementing Ridge Regression": [[17, "deriving-and-implementing-ridge-regression"]], "Deriving the Lasso Regression Equations": [[29, "deriving-the-lasso-regression-equations"], [30, "deriving-the-lasso-regression-equations"], [30, "id6"]], "Deriving the Ridge Regression Equations": [[29, "deriving-the-ridge-regression-equations"], [30, "deriving-the-ridge-regression-equations"], [30, "id3"]], "Deriving the back propagation code for a multilayer perceptron model": [[12, "deriving-the-back-propagation-code-for-a-multilayer-perceptron-model"]], "Developing a code for doing neural networks with back propagation": [[1, "developing-a-code-for-doing-neural-networks-with-back-propagation"]], "Diagonalize the sample covariance matrix to obtain the principal components": [[11, "diagonalize-the-sample-covariance-matrix-to-obtain-the-principal-components"]], "Different kernels and Mercer\u2019s theorem": [[8, "different-kernels-and-mercer-s-theorem"]], "Disadvantages": [[9, "disadvantages"]], "Discriminative Modeling": [[28, "discriminative-modeling"]], "Domains and probabilities": [[25, "domains-and-probabilities"]], "Dropout": [[1, "dropout"]], "Economy-size SVD": [[29, "economy-size-svd"], [30, "economy-size-svd"]], "Elements of Probability Theory and Statistical Data Analysis": [[25, null]], "Empirical Evidence: Convergence Time and Memory in Practice": [[31, "empirical-evidence-convergence-time-and-memory-in-practice"]], "Ensemble Methods: From a Single Tree to Many Trees and Extreme Boosting, Meet the Jungle of Methods": [[10, null]], "Entropy and the ID3 algorithm": [[9, "entropy-and-the-id3-algorithm"]], "Essential elements of ML": [[28, "essential-elements-of-ml"]], "Evaluate model performance on test data": [[1, "evaluate-model-performance-on-test-data"]], "Example 2": [[29, "example-2"]], "Example 3": [[29, "example-3"]], "Example 4": [[29, "example-4"]], "Example Matrix": [[29, "example-matrix"], [30, "example-matrix"]], "Example code for Bias-Variance tradeoff": [[32, "example-code-for-bias-variance-tradeoff"]], "Example of discriminative modeling, taken from Generative Deep Learning by David Foster": [[28, "example-of-discriminative-modeling-taken-from-generative-deep-learning-by-david-foster"]], "Example of generative modeling, taken from Generative Deep Learning by David Foster": [[28, "example-of-generative-modeling-taken-from-generative-deep-learning-by-david-foster"]], "Example of own Standard scaling": [[29, "example-of-own-standard-scaling"]], "Example relevant for the exercises": [[29, "example-relevant-for-the-exercises"]], "Example: Exponential decay": [[2, "example-exponential-decay"]], "Example: Population growth": [[2, "example-population-growth"]], "Example: The diffusion equation": [[2, "example-the-diffusion-equation"]], "Example: binary classification problem": [[1, "example-binary-classification-problem"]], "Examples": [[28, "examples"]], "Examples of likelihood functions used in logistic regression and neural networks": [[7, "examples-of-likelihood-functions-used-in-logistic-regression-and-neural-networks"]], "Exercise 1 - Choice of model and degrees of freedom": [[17, "exercise-1-choice-of-model-and-degrees-of-freedom"]], "Exercise 1 - Finding the derivative of Matrix-Vector expressions": [[16, "exercise-1-finding-the-derivative-of-matrix-vector-expressions"]], "Exercise 1 - Github Setup": [[15, "exercise-1-github-setup"]], "Exercise 1, scale your data": [[18, "exercise-1-scale-your-data"]], "Exercise 1: Creating the report document": [[20, "exercise-1-creating-the-report-document"]], "Exercise 1: Expectation values for ordinary least squares expressions": [[19, "exercise-1-expectation-values-for-ordinary-least-squares-expressions"]], "Exercise 1: Setting up various Python environments": [[0, "exercise-1-setting-up-various-python-environments"]], "Exercise 2 - Deriving the expression for OLS": [[16, "exercise-2-deriving-the-expression-for-ols"]], "Exercise 2 - Deriving the expression for Ridge Regression": [[17, "exercise-2-deriving-the-expression-for-ridge-regression"]], "Exercise 2 - Setting up a Github repository": [[15, "exercise-2-setting-up-a-github-repository"]], "Exercise 2, calculate the gradients": [[18, "exercise-2-calculate-the-gradients"]], "Exercise 2: Adding good figures": [[20, "exercise-2-adding-good-figures"]], "Exercise 2: Expectation values for Ridge regression": [[19, "exercise-2-expectation-values-for-ridge-regression"]], "Exercise 2: making your own data and exploring scikit-learn": [[0, "exercise-2-making-your-own-data-and-exploring-scikit-learn"]], "Exercise 3 - Creating feature matrix and implementing OLS using the analytical expression": [[16, "exercise-3-creating-feature-matrix-and-implementing-ols-using-the-analytical-expression"]], "Exercise 3 - Fitting an OLS model to data": [[15, "exercise-3-fitting-an-ols-model-to-data"]], "Exercise 3 - Scaling data": [[17, "exercise-3-scaling-data"]], "Exercise 3 - Setting up a Python virtual environment": [[15, "exercise-3-setting-up-a-python-virtual-environment"]], "Exercise 3, using the analytical formulae for OLS and Ridge regression to find the optimal paramters \\boldsymbol{\\theta}": [[18, "exercise-3-using-the-analytical-formulae-for-ols-and-ridge-regression-to-find-the-optimal-paramters-boldsymbol-theta"]], "Exercise 3: Deriving the expression for the Bias-Variance Trade-off": [[19, "exercise-3-deriving-the-expression-for-the-bias-variance-trade-off"]], "Exercise 3: Normalizing our data": [[0, "exercise-3-normalizing-our-data"]], "Exercise 3: Writing an abstract and introduction": [[20, "exercise-3-writing-an-abstract-and-introduction"]], "Exercise 4 - Fitting a polynomial": [[16, "exercise-4-fitting-a-polynomial"]], "Exercise 4 - Implementing Ridge Regression": [[17, "exercise-4-implementing-ridge-regression"]], "Exercise 4 - Testing multiple hyperparameters": [[17, "exercise-4-testing-multiple-hyperparameters"]], "Exercise 4 - The train-test split": [[15, "exercise-4-the-train-test-split"]], "Exercise 4, Implementing the simplest form for gradient descent": [[18, "exercise-4-implementing-the-simplest-form-for-gradient-descent"]], "Exercise 4: Adding Ridge Regression": [[0, "exercise-4-adding-ridge-regression"]], "Exercise 4: Computing the Bias and Variance": [[19, "exercise-4-computing-the-bias-and-variance"]], "Exercise 4: Making the code available and presentable": [[20, "exercise-4-making-the-code-available-and-presentable"]], "Exercise 5 - Comparing your code with sklearn": [[16, "exercise-5-comparing-your-code-with-sklearn"]], "Exercise 5, Ridge regression and a new Synthetic Dataset": [[18, "exercise-5-ridge-regression-and-a-new-synthetic-dataset"]], "Exercise 5: Analytical exercises": [[0, "exercise-5-analytical-exercises"]], "Exercise 5: Interpretation of scaling and metrics": [[19, "exercise-5-interpretation-of-scaling-and-metrics"]], "Exercise 5: Referencing": [[20, "exercise-5-referencing"]], "Exercise: Cross-validation as resampling techniques, adding more complexity": [[6, "exercise-cross-validation-as-resampling-techniques-adding-more-complexity"]], "Exercise: Analysis of real data": [[6, "exercise-analysis-of-real-data"]], "Exercise: Bias-variance trade-off and resampling techniques": [[6, "exercise-bias-variance-trade-off-and-resampling-techniques"]], "Exercise: Lasso Regression on the Franke function with resampling": [[6, "exercise-lasso-regression-on-the-franke-function-with-resampling"]], "Exercise: Ordinary Least Square (OLS) on the Franke function": [[6, "exercise-ordinary-least-square-ols-on-the-franke-function"]], "Exercise: Ridge Regression on the Franke function with resampling": [[6, "exercise-ridge-regression-on-the-franke-function-with-resampling"]], "Exercises": [[0, "exercises"]], "Exercises and Projects": [[6, "exercises-and-projects"]], "Exercises week 34": [[15, null]], "Exercises week 35": [[16, null]], "Exercises week 36": [[17, null]], "Exercises week 37": [[18, null]], "Exercises week 38": [[19, null]], "Exercises week 39": [[20, null]], "Expectation value and variance": [[32, "expectation-value-and-variance"]], "Expectation value and variance for \\boldsymbol{\\theta}": [[32, "expectation-value-and-variance-for-boldsymbol-theta"]], "Expectation values": [[25, "expectation-values"]], "Extending to more than one variable": [[30, "extending-to-more-than-one-variable"]], "Extremely useful tools, strongly recommended": [[28, "extremely-useful-tools-strongly-recommended"]], "Feed-forward neural networks": [[12, "feed-forward-neural-networks"]], "Feed-forward pass": [[1, "feed-forward-pass"]], "Final back propagating equation": [[12, "final-back-propagating-equation"]], "Finding the Limit": [[32, "finding-the-limit"]], "Fine-tuning neural network hyperparameters": [[1, "fine-tuning-neural-network-hyperparameters"]], "Fitting an Equation of State for Dense Nuclear Matter": [[0, "fitting-an-equation-of-state-for-dense-nuclear-matter"]], "Fixing the singularity": [[29, "fixing-the-singularity"], [30, "fixing-the-singularity"]], "Format for electronic delivery of report and programs": [[23, "format-for-electronic-delivery-of-report-and-programs"]], "Frequently used scaling functions": [[29, "frequently-used-scaling-functions"], [31, "frequently-used-scaling-functions"]], "From OLS to Ridge and Lasso": [[30, "from-ols-to-ridge-and-lasso"]], "From one to many layers, the universal approximation theorem": [[12, "from-one-to-many-layers-the-universal-approximation-theorem"]], "Functionality in Scikit-Learn": [[29, "functionality-in-scikit-learn"], [31, "functionality-in-scikit-learn"]], "Further Dimensionality Remarks": [[3, "further-dimensionality-remarks"]], "Further properties (important for our analyses later)": [[5, "further-properties-important-for-our-analyses-later"], [29, "further-properties-important-for-our-analyses-later"], [30, "further-properties-important-for-our-analyses-later"]], "Gaussian Elimination": [[22, "gaussian-elimination"]], "General Features": [[9, "general-features"]], "General linear models and linear algebra": [[28, "general-linear-models-and-linear-algebra"]], "Generalizing the fitting procedure as a linear algebra problem": [[28, "generalizing-the-fitting-procedure-as-a-linear-algebra-problem"], [28, "id1"]], "Generative Adversarial Networks": [[4, "generative-adversarial-networks"]], "Generative Models": [[4, "generative-models"]], "Generative Versus Discriminative Modeling": [[28, "generative-versus-discriminative-modeling"]], "Geometric Interpretation and link with Singular Value Decomposition": [[11, "geometric-interpretation-and-link-with-singular-value-decomposition"]], "Getting started with project 1": [[20, "getting-started-with-project-1"]], "Gradient Boosting, Classification Example": [[10, "gradient-boosting-classification-example"]], "Gradient Boosting, Examples of Regression": [[10, "gradient-boosting-examples-of-regression"]], "Gradient Clipping": [[1, "gradient-clipping"]], "Gradient Descent Example": [[30, "id1"], [31, "id1"]], "Gradient boosting: Basics with Steepest Descent/Functional Gradient Descent": [[10, "gradient-boosting-basics-with-steepest-descent-functional-gradient-descent"]], "Gradient descent": [[2, "gradient-descent"]], "Gradient descent and Ridge": [[30, "gradient-descent-and-ridge"], [31, "gradient-descent-and-ridge"]], "Gradient descent and revisiting Ordinary Least Squares from last week": [[31, "gradient-descent-and-revisiting-ordinary-least-squares-from-last-week"]], "Gradient descent example": [[30, "gradient-descent-example"], [31, "gradient-descent-example"]], "Grading": [[26, "grading"], [26, "id2"], [28, "grading"]], "How to take derivatives of Matrix-Vector expressions": [[16, "how-to-take-derivatives-of-matrix-vector-expressions"]], "Hyperplanes and all that": [[8, "hyperplanes-and-all-that"]], "Identifying Terms": [[32, "identifying-terms"]], "Important Matrix and vector handling packages": [[22, "important-matrix-and-vector-handling-packages"]], "Important technicalities: More on Rescaling data": [[29, "important-technicalities-more-on-rescaling-data"]], "Improving gradient descent with momentum": [[31, "improving-gradient-descent-with-momentum"]], "Improving performance": [[1, "improving-performance"]], "In summary": [[26, "in-summary"]], "Including Stochastic Gradient Descent with Autograd": [[13, "including-stochastic-gradient-descent-with-autograd"], [31, "including-stochastic-gradient-descent-with-autograd"]], "Incremental PCA": [[11, "incremental-pca"]], "Independent and Identically Distributed (iid)": [[32, "independent-and-identically-distributed-iid"]], "Installing R, C++, cython or Julia": [[28, "installing-r-c-cython-or-julia"]], "Installing R, C++, cython, Numba etc": [[28, "installing-r-c-cython-numba-etc"]], "Instructor information": [[26, "instructor-information"]], "Interpretations and optimizing our parameters": [[28, "interpretations-and-optimizing-our-parameters"], [28, "id2"], [28, "id3"], [29, "interpretations-and-optimizing-our-parameters"], [29, "id1"], [29, "id2"]], "Interpreting the Ridge results": [[29, "interpreting-the-ridge-results"], [30, "interpreting-the-ridge-results"], [30, "id4"]], "Introducing JAX": [[13, "introducing-jax"]], "Introducing the Covariance and Correlation functions": [[11, "introducing-the-covariance-and-correlation-functions"], [29, "introducing-the-covariance-and-correlation-functions"]], "Introduction": [[0, "introduction"], [6, "introduction"], [21, "introduction"], [22, "introduction"]], "Introduction to numerical projects": [[23, "introduction-to-numerical-projects"]], "Iterative Fitting, Classification and AdaBoost": [[10, "iterative-fitting-classification-and-adaboost"]], "Iterative Fitting, Regression and Squared-error Cost Function": [[10, "iterative-fitting-regression-and-squared-error-cost-function"]], "Kernel PCA": [[11, "kernel-pca"]], "Kernels and non-linearity": [[8, "kernels-and-non-linearity"]], "LU Decomposition, the inverse of a matrix": [[22, "lu-decomposition-the-inverse-of-a-matrix"]], "Lasso Regression": [[30, "lasso-regression"]], "Lasso case": [[30, "lasso-case"]], "Layers": [[1, "layers"]], "Layers used to build CNNs": [[3, "layers-used-to-build-cnns"]], "Learning goals": [[15, "learning-goals"], [16, "learning-goals"], [17, "learning-goals"], [18, "learning-goals"], [19, "learning-goals"], [20, "learning-goals"]], "Learning outcomes": [[21, "learning-outcomes"], [28, "learning-outcomes"]], "Lectures and ComputerLab": [[28, "lectures-and-computerlab"]], "Limitations of supervised learning with deep networks": [[1, "limitations-of-supervised-learning-with-deep-networks"]], "Linear Algebra, Handling of Arrays and more Python Features": [[22, null]], "Linear Regression": [[0, null]], "Linear Regression Problems": [[29, "linear-regression-problems"], [30, "linear-regression-problems"]], "Linear Regression and the SVD": [[30, "linear-regression-and-the-svd"]], "Linear Regression, basic elements": [[0, "linear-regression-basic-elements"]], "Linking Bayes\u2019 Theorem with Ridge and Lasso Regression": [[5, "linking-bayes-theorem-with-ridge-and-lasso-regression"]], "Linking the regression analysis with a statistical interpretation": [[5, "linking-the-regression-analysis-with-a-statistical-interpretation"], [32, "linking-the-regression-analysis-with-a-statistical-interpretation"]], "Linking with the SVD": [[5, "linking-with-the-svd"], [29, "linking-with-the-svd"]], "Links to relevant courses at the University of Oslo": [[27, "links-to-relevant-courses-at-the-university-of-oslo"]], "Logistic Regression": [[7, null], [7, "id1"]], "MNIST and GANs": [[4, "mnist-and-gans"]], "Machine Learning": [[28, "machine-learning"]], "Machine learning": [[21, "machine-learning"]], "Main textbooks": [[28, "main-textbooks"]], "Making a tree": [[9, "making-a-tree"]], "Making your own Bootstrap: Changing the Level of the Decision Tree": [[10, "making-your-own-bootstrap-changing-the-level-of-the-decision-tree"]], "Making your own test-train splitting": [[29, "making-your-own-test-train-splitting"]], "Material for exercises week 35": [[29, "material-for-exercises-week-35"]], "Material for lab sessions sessions Tuesday and Wednesday": [[30, "material-for-lab-sessions-sessions-tuesday-and-wednesday"]], "Material for lecture Monday September 2": [[30, "material-for-lecture-monday-september-2"]], "Material for lecture Monday September 8": [[31, "material-for-lecture-monday-september-8"]], "Material for the lab sessions": [[31, "material-for-the-lab-sessions"], [32, "material-for-the-lab-sessions"]], "Mathematical Interpretation of Ordinary Least Squares": [[5, "mathematical-interpretation-of-ordinary-least-squares"], [29, "mathematical-interpretation-of-ordinary-least-squares"], [30, "mathematical-interpretation-of-ordinary-least-squares"]], "Mathematical optimization of convex functions": [[8, "mathematical-optimization-of-convex-functions"]], "Mathematics of CNNs": [[3, "mathematics-of-cnns"]], "Mathematics of the SVD and implications": [[5, "mathematics-of-the-svd-and-implications"], [29, "mathematics-of-the-svd-and-implications"], [30, "mathematics-of-the-svd-and-implications"]], "Matrices in Python": [[28, "matrices-in-python"]], "Matrix multiplication": [[1, "matrix-multiplication"]], "Matrix-vector notation and activation": [[12, "matrix-vector-notation-and-activation"]], "Maximum Likelihood Estimation (MLE)": [[32, "maximum-likelihood-estimation-mle"]], "Meet the covariance!": [[25, "meet-the-covariance"]], "Meet the Covariance Matrix": [[5, "meet-the-covariance-matrix"], [29, "meet-the-covariance-matrix"]], "Meet the Hessian Matrix": [[29, "meet-the-hessian-matrix"]], "Meet the Pandas": [[28, "meet-the-pandas"]], "Memory Usage and Scalability": [[31, "memory-usage-and-scalability"]], "Memory constraints": [[31, "memory-constraints"]], "Min-Max Scaling": [[29, "min-max-scaling"]], "Momentum based GD": [[13, "momentum-based-gd"], [31, "momentum-based-gd"]], "More complicated Example: The Ising model": [[6, "more-complicated-example-the-ising-model"]], "More examples on bootstrap and cross-validation and errors": [[32, "more-examples-on-bootstrap-and-cross-validation-and-errors"]], "More interpretations": [[29, "more-interpretations"], [30, "more-interpretations"], [30, "id5"]], "More on Dimensionalities": [[3, "more-on-dimensionalities"]], "More on Rescaling data": [[6, "more-on-rescaling-data"]], "More on Steepest descent": [[30, "more-on-steepest-descent"]], "More on convex functions": [[30, "more-on-convex-functions"]], "More preprocessing": [[29, "more-preprocessing"], [31, "more-preprocessing"]], "Motivation for Adaptive Step Sizes": [[31, "motivation-for-adaptive-step-sizes"]], "Multilayer perceptrons": [[12, "multilayer-perceptrons"]], "Network requirements": [[2, "network-requirements"]], "Neural Networks vs CNNs": [[3, "neural-networks-vs-cnns"]], "Neural networks": [[12, null]], "Non-Convex Problems": [[31, "non-convex-problems"]], "Note about SVD Calculations": [[29, "note-about-svd-calculations"], [30, "note-about-svd-calculations"]], "Note on Scikit-Learn": [[30, "note-on-scikit-learn"]], "Numerical experiments and the covariance, central limit theorem": [[25, "numerical-experiments-and-the-covariance-central-limit-theorem"]], "Numpy and arrays": [[22, "numpy-and-arrays"], [28, "numpy-and-arrays"]], "Numpy examples and Important Matrix and vector handling packages": [[28, "numpy-examples-and-important-matrix-and-vector-handling-packages"]], "Optimization and gradient descent, the central part of any Machine Learning algortithm": [[30, "optimization-and-gradient-descent-the-central-part-of-any-machine-learning-algortithm"]], "Optimization, the central part of any Machine Learning algortithm": [[13, null]], "Optimizing our parameters": [[28, "optimizing-our-parameters"]], "Optimizing our parameters, more details": [[28, "optimizing-our-parameters-more-details"]], "Optimizing the cost function": [[1, "optimizing-the-cost-function"]], "Organizing our data": [[0, "organizing-our-data"], [28, "organizing-our-data"]], "Other Matrix and Vector Operations": [[22, "other-matrix-and-vector-operations"]], "Other Types of Recurrent Neural Networks": [[4, "other-types-of-recurrent-neural-networks"]], "Other courses on Data science and Machine Learning at UiO": [[28, "other-courses-on-data-science-and-machine-learning-at-uio"]], "Other courses on Data science and Machine Learning at UiO, contn": [[28, "other-courses-on-data-science-and-machine-learning-at-uio-contn"]], "Other popular texts": [[28, "other-popular-texts"]], "Other techniques": [[11, "other-techniques"]], "Other types of networks": [[12, "other-types-of-networks"]], "Other ways of visualizing the trees": [[9, "other-ways-of-visualizing-the-trees"]], "Our model for the nuclear binding energies": [[28, "our-model-for-the-nuclear-binding-energies"]], "Overview of first week": [[28, "overview-of-first-week"]], "Overview video on Stochastic Gradient Descent (SGD)": [[31, "overview-video-on-stochastic-gradient-descent-sgd"]], "Own code for Ordinary Least Squares": [[28, "own-code-for-ordinary-least-squares"], [29, "own-code-for-ordinary-least-squares"]], "PCA and scikit-learn": [[11, "pca-and-scikit-learn"]], "Pandas AI": [[28, "pandas-ai"]], "Part a : Ordinary Least Square (OLS) for the Runge function": [[23, "part-a-ordinary-least-square-ols-for-the-runge-function"]], "Part b: Adding Ridge regression for the Runge function": [[23, "part-b-adding-ridge-regression-for-the-runge-function"]], "Part c: Writing your own gradient descent code": [[23, "part-c-writing-your-own-gradient-descent-code"]], "Part d: Including momentum and more advanced ways to update the learning the rate": [[23, "part-d-including-momentum-and-more-advanced-ways-to-update-the-learning-the-rate"]], "Part e: Writing our own code for Lasso regression": [[23, "part-e-writing-our-own-code-for-lasso-regression"]], "Part f: Stochastic gradient descent": [[23, "part-f-stochastic-gradient-descent"]], "Part g: Bias-variance trade-off and resampling techniques": [[23, "part-g-bias-variance-trade-off-and-resampling-techniques"]], "Part h): Cross-validation as resampling techniques, adding more complexity": [[23, "part-h-cross-validation-as-resampling-techniques-adding-more-complexity"]], "Partial Differential Equations": [[2, "partial-differential-equations"]], "Plans for week 35": [[29, "plans-for-week-35"]], "Plans for week 36": [[30, "plans-for-week-36"]], "Plans for week 37, lecture Monday": [[31, "plans-for-week-37-lecture-monday"]], "Plans for week 38, lecture Monday September 15": [[32, "plans-for-week-38-lecture-monday-september-15"]], "Plotting the Histogram": [[32, "plotting-the-histogram"]], "Practical tips": [[13, "practical-tips"], [31, "practical-tips"]], "Practicalities": [[26, "practicalities"], [26, "id1"]], "Preamble: Note on writing reports, using reference material, AI and other tools": [[23, "preamble-note-on-writing-reports-using-reference-material-ai-and-other-tools"]], "Predicting New Points With A Trained Recurrent Neural Network": [[4, "predicting-new-points-with-a-trained-recurrent-neural-network"]], "Preprocessing our data": [[29, "preprocessing-our-data"]], "Prerequisites": [[28, "prerequisites"]], "Prerequisites and background": [[21, "prerequisites-and-background"]], "Prerequisites: Collect and pre-process data": [[3, "prerequisites-collect-and-pre-process-data"]], "Probability Distribution Functions": [[25, "probability-distribution-functions"]], "Program example for gradient descent with Ridge Regression": [[30, "program-example-for-gradient-descent-with-ridge-regression"], [31, "program-example-for-gradient-descent-with-ridge-regression"]], "Program for stochastic gradient": [[13, "program-for-stochastic-gradient"]], "Project 1 on Machine Learning, deadline October 6 (midnight), 2025": [[23, null]], "Properties of PDFs": [[25, "properties-of-pdfs"]], "Pros and cons": [[31, "pros-and-cons"]], "Pros and cons of trees, pros": [[9, "pros-and-cons-of-trees-pros"]], "Python installers": [[21, "python-installers"], [28, "python-installers"]], "RMS prop": [[13, "rms-prop"]], "RMSProp algorithm, taken from Goodfellow et al": [[31, "rmsprop-algorithm-taken-from-goodfellow-et-al"]], "RMSProp: Adaptive Learning Rates": [[31, "rmsprop-adaptive-learning-rates"]], "RMSprop for adaptive learning rate with Stochastic Gradient Descent": [[31, "rmsprop-for-adaptive-learning-rate-with-stochastic-gradient-descent"]], "Random Numbers": [[25, "random-numbers"]], "Random forests": [[10, "random-forests"]], "Randomized PCA": [[11, "randomized-pca"]], "Reading material": [[28, "reading-material"]], "Reading recommendations:": [[29, "reading-recommendations"]], "Reading suggestions week 34": [[28, "reading-suggestions-week-34"]], "Readings and Videos": [[32, "readings-and-videos"]], "Readings and Videos:": [[31, "readings-and-videos"]], "Recurrent neural networks": [[12, "recurrent-neural-networks"]], "Recurrent neural networks: Overarching view": [[4, null]], "Reducing the number of degrees of freedom, overarching view": [[0, "reducing-the-number-of-degrees-of-freedom-overarching-view"], [29, "reducing-the-number-of-degrees-of-freedom-overarching-view"]], "Reformulating the problem": [[2, "reformulating-the-problem"]], "Regression Case": [[10, "regression-case"]], "Regression analysis and resampling methods": [[23, "regression-analysis-and-resampling-methods"]], "Regression analysis, overarching aims": [[28, "regression-analysis-overarching-aims"]], "Regression analysis, overarching aims II": [[28, "regression-analysis-overarching-aims-ii"]], "Regularization": [[1, "regularization"]], "Reminder from last week": [[29, "reminder-from-last-week"]], "Reminder on Newton-Raphson\u2019s method": [[30, "reminder-on-newton-raphson-s-method"]], "Reminder on Statistics": [[6, "reminder-on-statistics"]], "Reminder on different scaling methods": [[31, "reminder-on-different-scaling-methods"]], "Replace or not": [[13, "replace-or-not"], [31, "replace-or-not"]], "Required Technologies": [[21, "required-technologies"]], "Resampling Methods": [[6, null]], "Resampling and the Bias-Variance Trade-off": [[19, "resampling-and-the-bias-variance-trade-off"]], "Resampling approaches can be computationally expensive": [[32, "resampling-approaches-can-be-computationally-expensive"]], "Resampling methods": [[6, "id1"], [32, "resampling-methods"], [32, "id2"]], "Resampling methods: Bootstrap": [[32, "resampling-methods-bootstrap"]], "Resampling methods: Bootstrap approach": [[32, "resampling-methods-bootstrap-approach"]], "Resampling methods: Bootstrap background": [[32, "resampling-methods-bootstrap-background"]], "Resampling methods: Bootstrap steps": [[32, "resampling-methods-bootstrap-steps"]], "Resampling methods: More Bootstrap background": [[32, "resampling-methods-more-bootstrap-background"]], "Residual Error": [[29, "residual-error"], [30, "residual-error"]], "Resources on differential equations and deep learning": [[2, "resources-on-differential-equations-and-deep-learning"]], "Revisiting Ordinary Least Squares": [[30, "revisiting-ordinary-least-squares"]], "Revisiting our Linear Regression Solvers": [[13, "revisiting-our-linear-regression-solvers"]], "Rewriting the Covariance and/or Correlation Matrix": [[29, "rewriting-the-covariance-and-or-correlation-matrix"]], "Rewriting the \\delta-function": [[32, "rewriting-the-delta-function"]], "Rewriting the fitting procedure as a linear algebra problem": [[28, "rewriting-the-fitting-procedure-as-a-linear-algebra-problem"]], "Rewriting the fitting procedure as a linear algebra problem, more details": [[28, "rewriting-the-fitting-procedure-as-a-linear-algebra-problem-more-details"]], "Ridge Regression": [[30, "ridge-regression"]], "Ridge and LASSO Regression": [[29, "ridge-and-lasso-regression"], [30, "ridge-and-lasso-regression"], [30, "id2"]], "Ridge and Lasso Regression": [[5, null], [5, "id1"]], "SGD example": [[31, "sgd-example"]], "SGD vs Full-Batch GD: Convergence Speed and Memory Comparison": [[31, "sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison"]], "SVD analysis": [[30, "svd-analysis"]], "Same code but now with momentum gradient descent": [[13, "same-code-but-now-with-momentum-gradient-descent"], [31, "same-code-but-now-with-momentum-gradient-descent"], [31, "id3"], [31, "id4"]], "Schedule first week": [[28, "schedule-first-week"]], "Schematic Regression Procedure": [[9, "schematic-regression-procedure"]], "Second moment of the gradient": [[31, "second-moment-of-the-gradient"]], "September 15-19": [[19, "september-15-19"]], "Setting up the Back propagation algorithm": [[12, "setting-up-the-back-propagation-algorithm"]], "Setting up the Matrix to be inverted": [[29, "setting-up-the-matrix-to-be-inverted"], [30, "setting-up-the-matrix-to-be-inverted"]], "Setting up the network using Autograd; The full program": [[2, "setting-up-the-network-using-autograd-the-full-program"]], "Similar (second order function now) problem but now with AdaGrad": [[13, "similar-second-order-function-now-problem-but-now-with-adagrad"], [31, "similar-second-order-function-now-problem-but-now-with-adagrad"]], "Simple Python Code to read in Data and perform Classification": [[9, "simple-python-code-to-read-in-data-and-perform-classification"]], "Simple case": [[29, "simple-case"], [30, "simple-case"]], "Simple code for solving the above problem": [[30, "simple-code-for-solving-the-above-problem"]], "Simple example code": [[31, "simple-example-code"]], "Simple example to illustrate Ordinary Least Squares, Ridge and Lasso Regression": [[30, "simple-example-to-illustrate-ordinary-least-squares-ridge-and-lasso-regression"]], "Simple geometric interpretation": [[30, "simple-geometric-interpretation"]], "Simple linear regression model using scikit-learn": [[0, "simple-linear-regression-model-using-scikit-learn"], [28, "simple-linear-regression-model-using-scikit-learn"]], "Simple one-dimensional second-order polynomial": [[18, "simple-one-dimensional-second-order-polynomial"]], "Simple program": [[30, "simple-program"], [31, "simple-program"]], "Slightly different approach": [[31, "slightly-different-approach"]], "Sneaking in automatic differentiation using Autograd": [[31, "sneaking-in-automatic-differentiation-using-autograd"]], "Software and needed installations": [[23, "software-and-needed-installations"], [28, "software-and-needed-installations"]], "Solving Differential Equations with Deep Learning": [[2, null]], "Solving the one dimensional Poisson equation": [[2, "solving-the-one-dimensional-poisson-equation"]], "Solving the wave equation with Neural Networks": [[2, "solving-the-wave-equation-with-neural-networks"]], "Some famous Matrices": [[22, "some-famous-matrices"]], "Some simple problems": [[13, "some-simple-problems"], [30, "some-simple-problems"]], "Some useful matrix and vector expressions": [[29, "some-useful-matrix-and-vector-expressions"]], "Splitting our Data in Training and Test data": [[0, "splitting-our-data-in-training-and-test-data"], [29, "splitting-our-data-in-training-and-test-data"]], "Standard Approach based on the Normal Distribution": [[32, "standard-approach-based-on-the-normal-distribution"]], "Standard steepest descent": [[13, "standard-steepest-descent"]], "Statistical analysis": [[32, "statistical-analysis"]], "Statistical analysis and optimization of data": [[21, "statistical-analysis-and-optimization-of-data"], [28, "statistical-analysis-and-optimization-of-data"]], "Steepest descent": [[13, "steepest-descent"], [30, "steepest-descent"]], "Stochastic Gradient Descent": [[31, "stochastic-gradient-descent"]], "Stochastic Gradient Descent (SGD)": [[13, "stochastic-gradient-descent-sgd"], [31, "stochastic-gradient-descent-sgd"]], "Stochastic variables and the main concepts, the discrete case": [[25, "stochastic-variables-and-the-main-concepts-the-discrete-case"]], "Strongly Convex Case": [[31, "strongly-convex-case"]], "Summing up": [[32, "summing-up"]], "Support Vector Machines, overarching aims": [[8, null]], "Systematic reduction": [[3, "systematic-reduction"]], "Teachers": [[28, "teachers"]], "Teachers and Grading": [[26, null]], "Teaching Assistants Fall semester 2023": [[26, "teaching-assistants-fall-semester-2023"]], "Tentative deadllines for projects": [[26, "tentative-deadllines-for-projects"]], "Testing the Means Squared Error as function of Complexity": [[0, "testing-the-means-squared-error-as-function-of-complexity"], [29, "testing-the-means-squared-error-as-function-of-complexity"]], "Textbooks": [[27, null]], "The Algorithm before theorem": [[11, "the-algorithm-before-theorem"]], "The Breast Cancer Data, now with Keras": [[1, "the-breast-cancer-data-now-with-keras"]], "The CART algorithm for Classification": [[9, "the-cart-algorithm-for-classification"]], "The CART algorithm for Regression": [[9, "the-cart-algorithm-for-regression"]], "The CIFAR01 data set": [[3, "the-cifar01-data-set"]], "The Central Limit Theorem": [[32, "the-central-limit-theorem"]], "The Hessian matrix": [[30, "the-hessian-matrix"], [31, "the-hessian-matrix"]], "The Hessian matrix for Ridge Regression": [[30, "the-hessian-matrix-for-ridge-regression"], [31, "the-hessian-matrix-for-ridge-regression"]], "The Jacobian": [[29, "the-jacobian"]], "The MNIST dataset again": [[3, "the-mnist-dataset-again"]], "The OLS case": [[30, "the-ols-case"]], "The RELU function family": [[1, "the-relu-function-family"]], "The Ridge case": [[30, "the-ridge-case"]], "The SVD, a Fantastic Algorithm": [[29, "the-svd-a-fantastic-algorithm"], [30, "the-svd-a-fantastic-algorithm"]], "The Softmax function": [[1, "the-softmax-function"]], "The \\chi^2 function": [[0, "the-chi-2-function"], [28, "the-chi-2-function"], [28, "id4"], [28, "id5"], [28, "id6"], [28, "id7"], [28, "id8"]], "The bias-variance tradeoff": [[6, "the-bias-variance-tradeoff"], [32, "the-bias-variance-tradeoff"]], "The code for solving the ODE": [[2, "the-code-for-solving-the-ode"]], "The complete code with a simple data set": [[29, "the-complete-code-with-a-simple-data-set"]], "The cost/loss function": [[29, "the-cost-loss-function"]], "The course has two central parts": [[21, "the-course-has-two-central-parts"]], "The derivative of the cost/loss function": [[30, "the-derivative-of-the-cost-loss-function"], [31, "the-derivative-of-the-cost-loss-function"]], "The equations": [[30, "the-equations"]], "The equations for ordinary least squares": [[29, "the-equations-for-ordinary-least-squares"]], "The first Case": [[30, "the-first-case"]], "The gradient step": [[31, "the-gradient-step"]], "The ideal": [[30, "the-ideal"]], "The logistic function": [[7, "the-logistic-function"]], "The mean squared error and its derivative": [[29, "the-mean-squared-error-and-its-derivative"]], "The moons example": [[8, "the-moons-example"]], "The multilayer perceptron (MLP)": [[12, "the-multilayer-perceptron-mlp"]], "The network with one input layer, specified number of hidden layers, and one output layer": [[2, "the-network-with-one-input-layer-specified-number-of-hidden-layers-and-one-output-layer"]], "The plethora of machine learning algorithms/methods": [[28, "the-plethora-of-machine-learning-algorithms-methods"]], "The same example but now with cross-validation": [[32, "the-same-example-but-now-with-cross-validation"]], "The sensitiveness of the gradient descent": [[30, "the-sensitiveness-of-the-gradient-descent"]], "The singular value decomposition": [[5, "the-singular-value-decomposition"], [29, "the-singular-value-decomposition"], [30, "the-singular-value-decomposition"]], "The two-dimensional case": [[8, "the-two-dimensional-case"]], "Theoretical Convergence Speed and convex optimization": [[31, "theoretical-convergence-speed-and-convex-optimization"]], "Time decay rate": [[31, "time-decay-rate"]], "To our real data: nuclear binding energies. Brief reminder on masses and binding energies": [[28, "to-our-real-data-nuclear-binding-energies-brief-reminder-on-masses-and-binding-energies"]], "Topics covered in this course: Statistical analysis and optimization of data": [[28, "topics-covered-in-this-course-statistical-analysis-and-optimization-of-data"]], "Towards the PCA theorem": [[11, "towards-the-pca-theorem"]], "Train and test datasets": [[1, "train-and-test-datasets"]], "Two-dimensional Objects": [[3, "two-dimensional-objects"]], "Type of problem": [[2, "type-of-problem"]], "Types of Machine Learning": [[28, "types-of-machine-learning"]], "Understanding what happens": [[32, "understanding-what-happens"]], "Use the books!": [[19, "use-the-books"]], "Useful Python libraries": [[21, "useful-python-libraries"], [28, "useful-python-libraries"]], "Using Autograd": [[13, "using-autograd"]], "Using forward Euler to solve the ODE": [[2, "using-forward-euler-to-solve-the-ode"]], "Using gradient descent methods, limitations": [[13, "using-gradient-descent-methods-limitations"], [30, "using-gradient-descent-methods-limitations"], [31, "using-gradient-descent-methods-limitations"]], "Various steps in cross-validation": [[32, "various-steps-in-cross-validation"]], "Visualization": [[1, "visualization"], [1, "id1"]], "Visualizing the Tree, Classification": [[9, "visualizing-the-tree-classification"]], "Week 34: Introduction to the course, Logistics and Practicalities": [[28, null]], "Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression": [[29, null]], "Week 36: Linear Regression and Gradient descent": [[30, null]], "Week 37: Gradient descent methods": [[31, null]], "Week 38: Statistical analysis, bias-variance tradeoff and resampling methods": [[32, null]], "What Is Generative Modeling?": [[28, "what-is-generative-modeling"]], "What does it mean?": [[29, "what-does-it-mean"], [30, "what-does-it-mean"]], "What is Machine Learning?": [[0, "what-is-machine-learning"]], "What is a good model?": [[0, "what-is-a-good-model"], [28, "what-is-a-good-model"]], "What is a good model? Can we define it?": [[28, "what-is-a-good-model-can-we-define-it"]], "When do we stop?": [[31, "when-do-we-stop"]], "Which activation function should I use?": [[1, "which-activation-function-should-i-use"]], "Why Combine Momentum and RMSProp?": [[31, "why-combine-momentum-and-rmsprop"]], "Why Linear Regression (aka Ordinary Least Squares and family)": [[28, "why-linear-regression-aka-ordinary-least-squares-and-family"]], "Why resampling methods": [[32, "why-resampling-methods"]], "Why resampling methods ?": [[32, "id1"]], "Wisconsin Cancer Data": [[7, "wisconsin-cancer-data"]], "With Lasso Regression": [[30, "with-lasso-regression"]], "Wrapping it up": [[32, "wrapping-it-up"]], "Writing Our First Generative Adversarial Network": [[4, "writing-our-first-generative-adversarial-network"]], "Writing our own PCA code": [[11, "writing-our-own-pca-code"]], "Writing the Cost Function": [[30, "writing-the-cost-function"]], "XGBoost: Extreme Gradient Boosting": [[10, "xgboost-extreme-gradient-boosting"]], "Yet another Example": [[30, "yet-another-example"]], "a) Expression for Ridge regression": [[17, "a-expression-for-ridge-regression"]], "scikit-learn implementation": [[1, "scikit-learn-implementation"]]}, "docnames": ["chapter1", "chapter10", "chapter11", "chapter12", "chapter13", "chapter2", "chapter3", "chapter4", "chapter5", "chapter6", "chapter7", "chapter8", "chapter9", "chapteroptimization", "clustering", "exercisesweek34", "exercisesweek35", "exercisesweek36", "exercisesweek37", "exercisesweek38", "exercisesweek39", "intro", "linalg", "project1", "schedule", "statistics", "teachers", "textbooks", "week34", "week35", "week36", "week37", "week38"], "envversion": {"sphinx": 62, "sphinx.domains.c": 3, "sphinx.domains.changeset": 1, "sphinx.domains.citation": 1, "sphinx.domains.cpp": 9, "sphinx.domains.index": 1, "sphinx.domains.javascript": 3, "sphinx.domains.math": 2, "sphinx.domains.python": 4, "sphinx.domains.rst": 2, "sphinx.domains.std": 2, "sphinx.ext.intersphinx": 1}, "filenames": ["chapter1.ipynb", "chapter10.ipynb", "chapter11.ipynb", "chapter12.ipynb", "chapter13.ipynb", "chapter2.ipynb", "chapter3.ipynb", "chapter4.ipynb", "chapter5.ipynb", "chapter6.ipynb", "chapter7.ipynb", "chapter8.ipynb", "chapter9.ipynb", "chapteroptimization.ipynb", "clustering.ipynb", "exercisesweek34.ipynb", "exercisesweek35.ipynb", "exercisesweek36.ipynb", "exercisesweek37.ipynb", "exercisesweek38.ipynb", "exercisesweek39.ipynb", "intro.md", "linalg.ipynb", "project1.ipynb", "schedule.md", "statistics.ipynb", "teachers.md", "textbooks.md", "week34.ipynb", "week35.ipynb", "week36.ipynb", "week37.ipynb", "week38.ipynb"], "indexentries": {}, "objects": {}, "objnames": {}, "objtypes": {}, "terms": {"": [0, 1, 2, 3, 4, 5, 6, 7, 9, 11, 12, 13, 15, 16, 17, 19, 21, 22, 23, 25, 26, 28, 29], "0": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 26, 28, 29, 30, 31, 32], "00": [0, 1, 5, 11, 28, 29], "000": [1, 3], "000000": [], "00000000e": [], "001": [2, 8, 13, 30, 31], "004": 5, "004113634617443131": 29, "004113634617443139": 29, "00411363461744314": 29, "004113634617443147": 29, "005b82": [], "00622f": [], "00727646693": [0, 28], "0072b2": [], "00749c": [], "008561": [], "0086649156": [0, 28], "00e0e0": [], "01": [0, 1, 2, 5, 9, 11, 13, 17, 27, 28, 29, 31], "010726": [], "0110": 25, "01719003e": [], "02": [0, 4, 7, 12, 28], "02334824": [], "023b95": [], "024c1a": [], "02857": 4, "02f": 6, "03077640549": 4, "03097597e": [], "031": 5, "04": 11, "0458": 9, "05": [4, 6], "0550ae": [], "062292565": 4, "062435": [], "06730814": [], "07": [], "0713": [0, 28], "07285": 3, "08": 25, "08078025e": [], "080808": [], "08336233266": 4, "08376632": 29, "083766322923899": 29, "0837663229239043": 29, "0917": 9, "0969da4a": [], "0d1117": [], "0n": [0, 28], "0x113e21950": 17, "1": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 24, 25, 26, 27, 28, 30, 31, 32], "10": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 18, 19, 22, 24, 25, 26, 28, 29, 30, 31, 32], "100": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 25, 26, 28, 29, 30, 31, 32], "1000": [0, 1, 2, 4, 5, 8, 11, 13, 14, 18, 19, 21, 25, 28, 30, 31], "10000": [2, 5, 6, 10, 11, 13, 25, 32], "100000": 8, "10001": 10, "1001": 25, "1002": 25, "1003": 25, "1005": 25, "1007": 32, "1009": 25, "101": 16, "1011": 25, "1013": 25, "1013904243": 25, "1015": 25, "102": 16, "1023": 25, "1024": 3, "1026": 25, "1027": 25, "103": 1, "1030": 25, "1037": 25, "1038": 25, "1040": 25, "1047": 25, "107": 16, "108": [], "10th": 9, "10x": [0, 28], "11": [0, 2, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 22, 23, 25, 27, 28, 29, 30, 31, 32], "110": [], "1100": 25, "1101": 25, "111": [1, 7, 12], "112": 16, "11340253": [], "11590451": [], "116": 16, "116329": [], "116633": [], "117": 16, "118": 16, "12": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 18, 22, 25, 27, 28, 29, 30, 31, 32], "120": 3, "121": [8, 9, 10, 16], "1215pm": [26, 28], "122": [8, 9, 10], "124": [0, 28], "125": 16, "127": [4, 16], "128": [3, 4, 13, 31], "129": 16, "1298": 9, "12pm": [26, 28], "13": [0, 2, 9, 12, 22, 25, 28], "131": 16, "133": 7, "135": 16, "136": 16, "14": [0, 2, 4, 6, 8, 9, 10, 12, 22, 25, 27, 29, 32], "141": 16, "1412": 31, "141414": [], "143": 16, "1446729567": 4, "149": 16, "14g": [6, 32], "15": [0, 2, 4, 6, 7, 8, 9, 12, 13, 23, 25, 28, 30, 31], "150": [4, 8], "152": 16, "153760": [], "156": 16, "157": [], "158": [], "159": 16, "15g": [6, 32], "15pm": 28, "16": [1, 2, 3, 4, 5, 8, 9, 10, 25, 28, 30, 32], "160": 16, "1603": 3, "161": 16, "162": 16, "16231451": 4, "163": 16, "16384": 3, "164": 16, "167": 16, "17": [1, 2, 8, 25], "172": 16, "173": 16, "175": 32, "176": 16, "178": 16, "179": 16, "1797": 1, "18": [2, 6, 7, 8, 9, 10, 25, 28, 32], "1807": 4, "181036": [], "18392847": [], "18c1c4": [], "19": [2, 25, 28, 32], "192": 32, "1940": [], "1943": 12, "19569961": 29, "19680801": [], "1970": [22, 28], "1973": 9, "1979": [6, 32], "1_1": 12, "1_2": 12, "1_3": 12, "1cm": [0, 8, 10, 25, 28], "1d": [1, 2, 3], "1e": [2, 4, 13, 14, 31], "1e10": 14, "1e1e1": [], "1e4": 6, "1f": 1, "1k": 22, "1n": [0, 28], "1x": [0, 28], "2": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 25, 27, 31, 32], "20": [0, 1, 2, 6, 7, 8, 16, 17, 25, 26, 28, 29, 30, 31, 32], "200": [0, 2, 3, 4, 8, 9, 10], "2000": [0, 29], "2001": [], "2004": [13, 30], "2006": 27, "2007": [], "20072279": [], "2008": [28, 31], "2009": [], "2010": 1, "2011": [1, 31], "2012": 31, "2013": [], "2014": [4, 31], "2015": 1, "2016": [0, 28], "2018": [0, 6, 29, 32], "2019": [], "2020": [], "2021": [6, 14, 29, 31], "2022": 28, "2024": 32, "2025": [18, 28, 29, 30, 31, 32], "21": [0, 1, 5, 7, 9, 12, 22, 28, 29, 30], "2116753732": 4, "215pm": [26, 28], "2167072": [], "22": [0, 1, 5, 12, 13, 22, 28, 29, 30], "221": 8, "225": 4, "22948497": [], "23": [1, 12, 22], "24": [0, 1, 22, 28], "242424": [], "24292f": [], "25": [2, 3, 4, 5, 6, 8, 9, 11, 29], "250": [2, 4, 7, 9], "25000": [], "250154": [], "252124": [], "253775": [], "255": 3, "256": [4, 31], "25x": 23, "26": [], "26303845": [], "264": [], "265": [], "265109911": 4, "266": [], "269": [], "27": 1, "270": [], "278": [30, 31], "27n_": 25, "28": [1, 3, 4], "283": [30, 31], "2830637392": 4, "2861": 25, "2873": 9, "2882": 25, "2886": 25, "2890": [0, 28], "2892": 25, "29": 29, "2915": 25, "2931": 28, "29364655": [], "294399745619595": [], "296247": [], "2968": 28, "2980": 28, "298273": [], "298375": [], "2990": 28, "2_": 12, "2_1": 12, "2_2": 12, "2_3": 12, "2_i": 12, "2_m": [6, 25, 32], "2_t": 13, "2_x": 25, "2a": 17, "2a1968": [], "2b": 25, "2b2b2b": [], "2c8f433990d1": 31, "2cm": 8, "2d": [1, 3, 11, 12, 21, 28], "2e": [6, 32], "2f": [0, 7, 9, 10, 11, 12, 28], "2g": 2, "2g_i": 2, "2k": 3, "2m": [6, 32], "2mvizaqfst8": 29, "2n": [0, 2, 3, 28, 29], "2nd": 9, "2p": 25, "2pt": 4, "2x": [0, 3, 8, 13, 28], "2x_ix_jy_iy_j": 8, "2x_j": 8, "2y_i": 10, "2y_j": 8, "3": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 24, 25, 26, 28, 30, 31, 32], "30": [0, 1, 4, 6, 7, 10, 13, 26, 31, 32], "30000": [0, 28], "3072": 3, "31": [12, 22, 25], "315": [6, 29, 31], "3155": [0, 5, 6, 29, 30, 31, 32], "32": [3, 4, 6, 12, 13, 22, 25, 31], "3200": 1, "3250": 1, "3297": [], "33": [12, 22, 26], "3303": [], "3310": [], "332331": [], "333": 7, "3331": [], "3337": [], "34": 22, "3436": [0, 28], "3437": [0, 28], "35": [0, 6, 23, 28, 30, 31], "3581341341": 4, "359": [5, 30], "36": [0, 5, 6, 18, 23, 25], "37": [23, 30, 32], "370782966": 4, "38": [23, 25], "387": 32, "39": [0, 26, 28], "3d": [2, 3, 4, 6, 13, 16, 32], "3d73a9": [], "3f": [1, 3, 9], "3n": 22, "3x": [2, 8], "3x_i": 2, "3y": 8, "4": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 23, 25, 28, 30, 31, 32], "40": [1, 6, 26, 28, 32], "400": 4, "4000": 28, "40008b9a5380fcacce3976bf7c08af5b": 31, "4050": [27, 28], "41": 22, "4155": [2, 15], "41589548": [], "42": [1, 4, 8, 9, 10, 22], "43": [0, 7, 22], "4310": 28, "436462435": 4, "437a6b": [], "44": [0, 22, 30, 31], "45": [26, 28], "46": [26, 28], "462": 7, "47": [26, 28], "473d18": [], "479465113": 4, "47958494": [], "48": [], "48257387": [26, 28], "49": [5, 6, 11], "49152": 3, "4940954": [0, 28], "4990": 25, "4992": 25, "4997": 25, "4c4b4be8": [], "4c4c7f": [9, 10], "4d": 3, "4f": 6, "4pm": [26, 28], "4y": 8, "4y_i": 10, "5": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 22, 23, 25, 28, 29, 30, 31, 32], "50": [1, 2, 3, 4, 6, 7, 8, 10, 13, 28, 29, 31, 32], "500": [1, 3, 4, 6, 9, 10, 13, 31, 32], "5018": 25, "506": [], "507d50": [9, 10], "50j": 13, "50x10": 1, "51": 10, "510": 1, "512132": [], "515151": [], "5177783846": 4, "53": 9, "5391cf": [], "54": [6, 25], "5411205": [], "54894451": [], "55": 1, "56": 1, "56536": [0, 28], "569": 1, "57": [0, 8, 26, 28], "571": [5, 30], "576": 32, "58": [10, 26, 28], "58a6ff70": [], "591317992": 4, "5ca7e4": [], "5cm": 25, "5f": [8, 31], "5x": [8, 18], "5y": 8, "6": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 18, 22, 25, 26, 28, 29, 30, 31, 32], "60": [1, 3], "60000": 4, "6019067271": 4, "606439": [], "622cbc": [], "625": 7, "63": 1, "64": [1, 3, 4, 13, 22, 28, 31], "64x50": 1, "65": [1, 8, 9], "66666691": [], "66707b": [], "66ccee": [], "66e9ec": [], "6730c5": [], "6887363571": 4, "69": [16, 25], "69069n_": 25, "691": [], "6980": 31, "6e7681": [], "6e7781": [], "6f98b3": [], "6n_": 25, "7": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 22, 23, 25, 27, 28, 29, 31, 32], "70": [1, 7], "702c00": [], "70653767": 4, "71": 1, "724": 3, "72f088": [], "73": [], "7304881": [], "737373": [], "75": [5, 6, 8, 11, 32], "76": [26, 28], "765": 7, "77": [26, 28], "7718": 9, "7782028952": 4, "77893972": [], "78": [], "797979": [], "7998f2": [], "79c0ff": [], "7d7d58": [9, 10], "7ee787": [], "7f4707": [], "8": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 18, 19, 22, 25, 26, 28, 30], "80": [0, 1, 5, 8, 17, 29], "800": [4, 7], "8045e5": [], "81": 1, "815am": [26, 28], "81b19b": [], "8250df": [], "84858": 32, "85": 1, "8702784034": 4, "8786ac": [], "88": 28, "8a4600": [], "8b949e": [], "8c8c8c": [], "8f": [6, 32], "8g": [6, 32], "8n": 22, "8x8": 1, "9": [0, 1, 2, 4, 5, 6, 7, 8, 9, 11, 12, 13, 22, 25, 28, 31], "90": 1, "9040": 9, "91": [26, 28], "912583": [], "91cbff": [], "92": [26, 28], "93": 16, "931": [0, 28], "933": [5, 30], "937": 25, "938": 25, "939": [0, 25, 28], "94": 25, "95": [1, 11, 32], "953800": [], "954": 25, "955820c21e8b": 4, "96": [6, 32], "960": 25, "961": 25, "962": 25, "9649652536": 4, "96611194e": [], "974eb7": [], "978": 32, "9780387310732": 27, "9780387848570": 27, "9781098134174": 28, "9781492032632": 27, "9781801819312": 28, "97898392": 29, "98": [0, 1, 16], "985": 25, "986": 25, "98661b": [], "989": 25, "9898ff": [9, 10], "99": [13, 16, 31, 32], "991": 25, "992": 25, "993": 25, "996": 5, "996b00": [], "999": [9, 25, 31], "999999": [], "9e86c8": [], "9e8741": [], "9f4e55": [], "9x": 6, "9y": 6, "A": [2, 3, 5, 6, 7, 10, 11, 12, 13, 15, 16, 19, 20, 21, 22, 24, 25, 26, 27, 29, 30, 31], "AND": 2, "AS": [], "AT": [], "And": [0, 3, 4, 5, 6, 9, 13, 20, 21, 23, 25, 30], "As": [0, 1, 2, 3, 4, 5, 6, 8, 10, 12, 13, 15, 16, 22, 23, 25, 28, 29, 30, 31, 32], "At": [0, 4, 6, 13, 20, 28, 31], "BE": [0, 28], "BUT": [], "BY": [], "Be": [2, 18, 21, 28], "Being": 13, "But": [0, 1, 2, 3, 5, 6, 9, 10, 16, 25, 29, 32], "By": [0, 3, 5, 6, 12, 13, 17, 19, 22, 28, 29, 30, 31, 32], "FOR": [], "For": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "IF": [6, 29, 31], "IN": 27, "If": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 18, 21, 22, 23, 25, 28, 29, 30, 31, 32], "In": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "Ising": [5, 12, 29, 30], "It": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "Its": [1, 2, 4, 11], "NO": [], "NOT": [], "No": [6, 9, 28, 29, 31], "Not": [0, 1, 5, 6, 29, 30, 31, 32], "OF": [], "ON": [], "OR": 25, "Of": 25, "On": [0, 3, 23, 25, 26, 27, 28, 31, 32], "One": [0, 1, 3, 4, 5, 6, 7, 8, 11, 12, 13, 17, 25, 29, 30, 31, 32], "Or": [0, 1, 6, 28], "SUCH": [], "Such": [0, 6, 12, 16, 25, 31, 32], "THE": [], "TO": [], "That": [0, 5, 7, 10, 11, 12, 14, 23, 25, 28, 32], "The": [4, 10, 13, 14, 16, 17, 18, 19, 20, 22, 23, 24, 25, 26, 27], "Then": [0, 1, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 22, 28, 30, 31, 32], "There": [0, 3, 4, 5, 6, 8, 9, 11, 12, 14, 15, 22, 23, 25, 26, 28, 29, 30, 31], "These": [0, 3, 4, 5, 8, 9, 10, 11, 12, 13, 14, 17, 18, 22, 23, 25, 26, 28, 29, 30, 31], "To": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 20, 22, 25, 29, 30, 31, 32], "WITH": [], "With": [0, 5, 6, 8, 9, 10, 11, 12, 14, 16, 19, 22, 23, 25, 28, 29, 32], "_": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 18, 19, 22, 23, 28, 29, 30, 31, 32], "_0": [5, 8, 10, 11, 13, 29, 30], "_1": [2, 5, 6, 8, 10, 11, 12, 13, 14, 22, 29, 30, 31], "_2": [2, 5, 8, 11, 12, 13, 22, 29, 31], "_3": 22, "_4": 22, "_9": [13, 31], "__array_finalize__": [], "__class__": 10, "__doc__": [6, 32], "__future__": [8, 9], "__getattribute__": [], "__import__": [], "__init__": 1, "__main__": 2, "__name__": [2, 10], "__new__": [], "__path__": [], "_auto1": [2, 3, 4, 5, 6, 7, 12, 13, 22, 25, 29, 30], "_auto10": [6, 12], "_auto11": 6, "_auto12": 6, "_auto2": [2, 3, 4, 5, 6, 12, 13, 22, 25], "_auto3": [3, 4, 5, 6, 12, 13, 22], "_auto4": [4, 6, 12, 13, 22], "_auto5": [4, 6, 12, 13, 22], "_auto6": [4, 6, 12, 22], "_auto7": [4, 6, 12, 22], "_auto8": [6, 12], "_auto9": [6, 12], "_build": [0, 21, 23, 27, 28], "_c": 1, "_center": [], "_compile_transl": [], "_compon": 11, "_data": [], "_depth": 9, "_export": [15, 16, 19], "_fraction": 9, "_i": [0, 1, 2, 5, 6, 7, 8, 11, 12, 13, 19, 23, 28, 29, 30, 31, 32], "_j": [0, 1, 2, 3, 5, 6, 8, 13, 19, 23, 29, 30, 31, 32], "_k": [13, 30, 31], "_l": 12, "_lambda": 6, "_leaf": 9, "_m": 10, "_mask": [], "_multilayer_perceptron": [], "_n": [2, 5, 8, 11, 13, 29, 30, 31], "_node": 9, "_norm": [], "_p": [5, 8, 29, 30], "_parse_numpydoc_see_also_sect": [], "_pydevd_bundl": [], "_ratio": 11, "_sampl": 9, "_split": [6, 9, 23], "_t": [13, 31], "_test": [6, 23], "_varianc": 11, "_weight": 9, "a0": 3, "a0111f": [], "a0faa0": [9, 10], "a1": [0, 28], "a11": [], "a12236": [], "a2": [0, 28], "a25e53": [], "a2bffc": [], "a3": [0, 28], "a4": [0, 28], "a5d6ff": [], "a_": [0, 1, 16, 22, 28, 29], "a_0": [0, 28], "a_1a": [0, 28], "a_2a": [0, 28], "a_3": [0, 28], "a_3a": [0, 28], "a_4": [0, 28], "a_4a": [0, 28], "a_h": 1, "a_i": [0, 1, 2, 12, 28], "a_j": [1, 12], "a_k": [0, 1, 12], "aa": [], "aaa": [], "aaron": 27, "ab": [0, 2, 5, 13, 14, 28, 29, 31], "ab6369": [], "ab_channel": 21, "abandon": 1, "abe338": [], "abid": 25, "abil": [0, 10], "abl": [0, 1, 4, 5, 6, 7, 10, 12, 13, 16, 18, 20, 23, 29, 30, 31], "about": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 19, 20, 21, 22, 23, 26, 31, 32], "abov": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 25, 27, 28, 29, 31, 32], "abovement": [6, 23, 28, 32], "abscissa": [13, 30], "absent": 31, "absolut": [0, 2, 5, 6, 13, 28, 29, 30, 32], "absorb": [29, 30], "abstract": [1, 31], "abund": 31, "ac": [], "acceler": [13, 31], "accept": [0, 3, 6, 9, 23, 29, 31], "access": [3, 11, 25, 28, 31], "accid": [4, 6, 32], "accompani": [0, 28, 29], "accomplish": [8, 9, 13, 31], "accord": [0, 1, 2, 5, 6, 9, 12, 13, 14, 25, 28, 30, 31, 32], "accordingli": 11, "account": [0, 3, 5, 13, 15, 16, 20, 25, 28, 31], "accumul": [12, 13, 25, 31], "accur": [0, 3, 4, 6, 10, 13, 31, 32], "accuraci": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 12, 28, 29, 30], "accuracy_scor": [0, 1, 10, 28], "accuracy_score_numpi": 1, "achiev": [0, 1, 5, 6, 8, 12, 22, 28, 31, 32], "aco": 25, "acquaint": 21, "acquir": [1, 21, 28], "acr": [], "across": [1, 3, 6, 9, 17, 21, 28, 32], "act": [1, 3, 22, 31], "action": 25, "activ": [0, 2, 3, 4, 9, 15, 24, 26, 28, 31], "activest": [], "actual": [0, 1, 4, 5, 6, 8, 11, 15, 16, 18, 22, 25, 28, 29, 30, 31, 32], "ad": [1, 3, 4, 5, 8, 13, 15, 16, 22, 30, 31, 32], "ada_clf": 10, "adaboostclassifi": 10, "adadelta": [13, 31], "adagrad": [23, 32], "adam": [1, 3, 4, 23, 28, 32], "adapt": [4, 6, 13, 17, 27, 30, 32], "add": [0, 1, 2, 3, 4, 5, 6, 8, 10, 11, 12, 15, 16, 17, 18, 20, 25, 26, 28, 29, 30, 31, 32], "add6ff": [], "add_": [], "add_subplot": [1, 7, 12, 14], "addendum": 5, "addeventlisten": [], "addit": [0, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 15, 21, 22, 23, 25, 26, 27, 28, 29, 32], "addition": [12, 13, 30, 31], "address": [1, 9, 11, 13, 28, 31], "adjac": [3, 12], "adjoint": [5, 29], "adjust": [0, 5, 12, 13, 30, 31], "admir": [0, 28], "advanc": [4, 6, 12, 27, 28, 31, 32], "advantag": [1, 3, 5, 6, 10, 13, 19, 22, 30, 31, 32], "adversari": 28, "advis": [], "afecionado": 28, "affect": [3, 15, 19], "affin": [0, 3, 8, 11, 29], "afford": 3, "aficionado": 28, "aforement": 14, "african": [], "after": [0, 1, 2, 4, 5, 6, 9, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "afterward": [0, 28], "ag": [0, 7, 28, 29], "ag_0": 2, "again": [0, 1, 4, 5, 6, 7, 8, 10, 11, 12, 13, 23, 25, 28, 29, 30, 32], "against": [1, 4, 7, 10], "agegroup": 7, "agegroupmean": 7, "aggreg": [9, 10, 31], "agorithm": 10, "agre": [5, 6, 25, 29, 30, 31, 32], "agreement": [13, 31], "ahead": 9, "ai": [0, 27], "aid": [11, 20, 31], "aim": [0, 1, 4, 6, 7, 11, 14, 16, 17, 19, 20, 21, 22, 23, 29, 32], "ainv": 5, "airplan": 3, "aka": 5, "al": [0, 2, 4, 16, 17, 20, 27, 28, 29, 30, 32], "alarm": [5, 7], "aldo": 29, "algebra": [0, 3, 5, 13, 21, 29, 30, 32], "algorithm": [0, 1, 2, 4, 5, 6, 7, 8, 13, 14, 16, 21, 22, 23, 25, 27, 32], "align": [0, 2, 5, 6, 7, 8, 13, 25, 28, 29, 30, 32], "all": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15, 18, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32], "allevi": [1, 13, 30], "alloc": [3, 22], "allow": [0, 1, 2, 3, 5, 6, 8, 10, 13, 15, 21, 22, 23, 28, 29, 30, 31, 32], "almost": [0, 1, 6, 8, 11, 13, 25, 30, 31, 32], "alon": [2, 9, 31], "along": [2, 3, 4, 5, 6, 9, 10, 11, 15, 20, 21, 22, 28, 29, 30, 32], "alpha": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 13, 14, 25, 28, 29, 30, 31, 32], "alpha_": [10, 31], "alpha_0": 3, "alpha_1": 3, "alpha_2": 3, "alpha_i": [3, 13], "alpha_k": 13, "alpha_m": 10, "alpha_n": 3, "alpha_opt": 13, "alreadi": [2, 3, 4, 5, 6, 10, 12, 15, 21, 22, 25, 28, 29, 30], "also": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "alter": 1, "altern": [0, 1, 4, 5, 6, 8, 9, 11, 13, 15, 18, 22, 23, 28, 29, 31, 32], "although": [0, 1, 5, 6, 8, 10, 13, 16, 19, 20, 28, 31, 32], "alwai": [0, 3, 5, 6, 12, 13, 16, 19, 23, 25, 28, 29, 30, 31, 32], "am": 4, "ame2016": [0, 28], "american": [], "amjith": [], "among": [0, 3, 5, 9, 10, 12, 22, 28, 29], "amongst": [5, 32], "amount": [0, 1, 3, 4, 6, 8, 10, 14, 21, 32], "an": [1, 2, 3, 5, 6, 7, 8, 9, 11, 12, 13, 14, 16, 17, 18, 19, 21, 22, 23, 25, 26, 27, 29, 30, 31, 32], "an_": 25, "anaconda": [0, 1, 21, 23, 28], "analogi": 13, "analys": [6, 32], "analysi": [1, 3, 4, 7, 14, 19, 22, 27, 31], "analyt": [2, 3, 5, 6, 7, 12, 13, 17, 21, 23, 28, 29, 30, 31, 32], "analyz": [0, 1, 3, 4, 5, 6, 16, 23, 25, 29, 30, 31], "andrew": 1, "angl": [0, 3, 9, 29, 31], "anharmon": 3, "ani": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 14, 15, 16, 19, 25, 28, 29, 31, 32], "anim": [4, 12], "ann": 12, "annot": [0, 1, 3, 7, 8, 28], "announc": 28, "anom": [], "anomali": [], "anonym": 18, "anoth": [0, 1, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 15, 22, 23, 25, 28, 29, 31], "ansatz": [0, 18, 28], "answer": [0, 1, 3, 5, 6, 19, 22, 23, 26, 28, 32], "antialias": [2, 6], "anticip": 4, "anymor": [1, 8], "anyon": [4, 8, 15], "anyth": [1, 15, 16, 25], "anytim": [26, 28], "anywai": [], "apach": 1, "apart": [11, 13, 30, 31], "api": [1, 21, 28], "appar": 2, "appear": [0, 1, 3, 13, 22, 25], "append": [1, 3, 4, 8, 9, 13, 19, 28, 31], "appendix": 23, "appli": [0, 1, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 18, 23, 25, 27, 28, 29, 31, 32], "applic": [0, 1, 3, 4, 5, 6, 7, 9, 12, 13, 16, 22, 25, 27, 28, 29, 30, 31, 32], "apply_gradi": 4, "approach": [1, 2, 4, 5, 6, 9, 10, 11, 12, 13, 15, 16, 18, 21, 23, 25, 27, 29, 30], "approch": 23, "appropri": [2, 6, 9, 12, 13, 17, 21, 25, 31, 32], "approv": 28, "approx": [0, 2, 3, 6, 10, 11, 13, 18, 23, 25, 28, 30, 31, 32], "approxim": [0, 1, 2, 3, 4, 5, 6, 7, 10, 11, 13, 19, 23, 25, 28, 29, 30, 31, 32], "apt": [0, 21, 23, 28], "aq": 25, "ar": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32], "aragorn": 28, "arang": [1, 3, 4, 6, 7, 9, 10, 12, 13, 28, 31], "arbitrari": [1, 4, 6, 8, 12, 13, 25, 30, 32], "arbitrarili": [0, 1, 11, 28, 31], "arc": 6, "architectur": [3, 4, 12], "archiv": 23, "area": [0, 3, 6, 27, 28], "argmax": [1, 11], "argmin": [4, 10, 14], "argsort": 11, "argu": [1, 13], "arguement": 19, "argument": [0, 2, 3, 5, 11, 12, 13, 17, 28, 29, 31, 32], "aris": [0, 6, 12, 13, 25, 28, 30, 32], "arithmet": [0, 13, 22, 28], "arm": [6, 29, 31], "armadillo": 22, "armin": [], "around": [0, 1, 4, 5, 6, 11, 18, 23, 25, 28, 32], "arrai": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 12, 13, 14, 16, 18, 21, 23, 25, 29, 30, 31, 32], "arrang": [3, 28], "arraybox": 13, "arriv": [0, 6, 9, 11, 22, 25, 28, 32], "arrow": 12, "arrowprop": 8, "art": [0, 1, 21], "articl": [0, 3, 4, 6, 10, 19, 28, 29, 30, 31, 32], "artifici": [0, 2, 7, 12, 27, 28], "artificialneuron": 12, "arug": 13, "arxiv": [3, 4, 31], "asarrai": [0, 6, 9, 29, 31], "asid": 29, "ask": [5, 6, 11, 12, 15, 19, 23, 32], "aspect": [0, 6, 21, 28, 29], "assembl": 3, "assembli": [0, 28], "assert": 4, "assess": [0, 6, 23, 28, 29, 32], "asset": [], "assici": 4, "assign": [0, 7, 8, 9, 12, 13, 14, 15, 24, 26, 27, 28], "associ": [0, 6, 9, 12, 14, 25, 28, 32], "assum": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "assumpt": [0, 3, 5, 6, 9, 11, 25, 28, 29], "ast": [0, 5, 6, 28, 32], "astyp": [4, 9, 10], "asymmetri": [0, 28], "asymptot": [4, 6, 31, 32], "atom": [0, 28], "attain": 31, "attempt": [0, 4, 6, 7, 8, 10, 28, 29, 31], "attend": 28, "attent": [0, 22, 28], "attract": [0, 10, 28], "attribut": [0, 9, 28], "audi": [0, 28], "audio": [3, 4], "august": [28, 29], "aurelien": [0, 27, 28], "austfjel": 6, "auth": 15, "authent": 15, "author": [0, 1, 10, 25], "authour": 28, "auto": [9, 10, 25], "auto_exampl": [23, 29], "autocor": 25, "autocorrelation_tim": 25, "autocorrelform": 25, "autocovari": 25, "autoencod": [4, 21, 28], "autoencond": 21, "autograd": [21, 28], "autom": [0, 21, 27, 28], "automac": 22, "automag": 28, "automat": [0, 1, 2, 3, 4, 11, 16, 21, 22, 28], "automobil": 3, "autonom": 4, "avail": [0, 1, 4, 6, 10, 11, 21, 22, 23, 24, 26, 27, 28, 32], "avali": 20, "averag": [0, 1, 3, 6, 9, 10, 13, 14, 25, 26, 28, 29, 32], "avoid": [0, 4, 5, 6, 9, 11, 13, 18, 22, 29, 31, 32], "awai": [2, 3, 6, 29, 31], "awar": [2, 10], "award": [26, 28], "ax": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 20, 22, 23, 28, 32], "axes3d": [2, 6, 13, 30, 31], "axes_grid1": 6, "axhlin": 8, "axi": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18, 22, 25, 28, 29, 30, 31, 32], "axiom": 5, "axvlin": [4, 8], "axvspan": 4, "b": [0, 1, 3, 4, 5, 6, 8, 9, 10, 12, 13, 14, 15, 16, 17, 19, 20, 25, 26, 28, 29, 30, 31, 32], "b1": 8, "b19db4": [], "b1bac4": [], "b2": 8, "b3": 8, "b35900": [], "b89784": [], "b_": [0, 1, 22], "b_0": 0, "b_1": [0, 2, 12, 13, 31], "b_2": [0, 13], "b_5": [13, 31], "b_group": 9, "b_i": [0, 1, 2, 12, 28], "b_ia_": [0, 28], "b_ia_i": 0, "b_index": 9, "b_j": [1, 12], "b_k": [0, 1, 12, 13, 31], "b_m": 12, "b_score": 9, "b_valu": 9, "ba": 31, "babcock": 28, "bach": 31, "bachelor": [24, 26], "back": [0, 3, 4, 5, 6, 8, 9, 10, 15, 16, 22, 25, 28, 31], "backbon": 22, "backend": [1, 4], "background": [27, 28], "backpropag": [1, 31], "backslash": [], "backtrack": 9, "backup": 22, "backward": [1, 2, 4, 12, 22, 31], "bad": [6, 17, 29], "badli": 25, "bag": [9, 21, 28], "bag_clf": 10, "baggin": 28, "baggingboot": 10, "baggingclassifi": 10, "baggingtre": 10, "bailei": [], "balanc": [6, 31, 32], "ballpark": 18, "band": 22, "bandwidth": 22, "banner": [], "bar": [0, 6, 11, 23, 28], "barber": 27, "bare": [4, 10], "base": [0, 1, 3, 4, 5, 7, 8, 9, 10, 14, 15, 16, 17, 21, 25, 26, 27, 28, 29, 30], "basi": [5, 7, 8, 10, 11, 12, 13, 22, 29, 30], "basic": [6, 8, 12, 13, 14, 15, 21, 23, 25, 28, 32], "basin": 31, "batch": [3, 4, 11, 12, 13, 30], "batch_shap": 4, "batch_siz": [1, 3, 4], "batchnorm": 4, "bay": 7, "bayesian": [5, 21, 27, 28], "bbbbbb": [], "beauti": [], "becam": [], "becaus": [0, 1, 2, 3, 4, 5, 6, 8, 9, 12, 13, 14, 28, 29, 30, 31, 32], "becom": [0, 1, 2, 5, 6, 7, 9, 12, 13, 19, 25, 28, 29, 30, 31, 32], "been": [0, 1, 2, 3, 4, 5, 6, 11, 12, 13, 20, 21, 22, 23, 28, 29, 31, 32], "befor": [0, 1, 2, 3, 4, 5, 6, 7, 8, 12, 13, 14, 16, 17, 18, 19, 20, 22, 23, 25, 28, 29, 31, 32], "beforehand": [0, 25, 28], "began": [], "begin": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 14, 15, 22, 25, 26, 28, 29, 30, 31, 32], "behav": [1, 6, 13, 30, 32], "behavior": [0, 1, 13, 28, 30, 31], "behaviour": [12, 31], "behind": [0, 1, 6, 8, 13, 28, 30], "being": [0, 1, 2, 3, 4, 5, 7, 8, 10, 11, 12, 13, 17, 20, 25, 28, 29, 30, 31], "believ": [9, 22], "belong": [7, 8, 9, 13, 14, 30], "below": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 18, 22, 23, 25, 28, 29, 30, 31, 32], "benchmark": 10, "benefici": [1, 13], "benefit": [0, 1, 4, 11, 13, 21, 28, 30, 31], "bengio": [1, 27, 28, 29, 31], "benign": [1, 7], "besid": [4, 5, 30], "bessel": [5, 29, 32], "best": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 15, 16, 18, 26, 28, 29, 30, 31, 32], "beta": [1, 3, 10, 11, 13, 16, 17, 19, 28, 29, 30], "beta1": [], "beta2": [], "beta_": [3, 13, 17, 29], "beta_0": [1, 3, 13, 29], "beta_1": [1, 3, 10, 13, 29, 31], "beta_1m_": 31, "beta_1x_i": 13, "beta_2": [3, 13, 31], "beta_2v_": 31, "beta_3": 3, "beta_i": [3, 31], "beta_j": [13, 29], "beta_k": 13, "beta_linreg": 13, "beta_m": 10, "beta_mg_m": 10, "beta_n": 3, "better": [0, 1, 2, 3, 4, 6, 9, 10, 11, 12, 13, 19, 20, 28, 29, 31, 32], "between": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 14, 15, 16, 17, 18, 19, 23, 25, 28, 29, 30, 31, 32], "beyond": [0, 1, 5, 6, 8, 13, 28, 29, 30, 31], "bf": [13, 14, 22, 25, 30], "bf5400": [], "bg": 28, "bgd": [13, 31], "bia": [0, 1, 2, 3, 5, 8, 9, 10, 12, 13, 20, 28, 29, 30], "bias": [1, 2, 3, 5, 6, 9, 12, 19, 31, 32], "bib": [], "bibliographi": 23, "bibtex": [], "big": [0, 1, 2, 5, 6, 14, 19, 31, 32], "bigger": [1, 6, 29], "bigr": 12, "bike": 9, "bilbo": 28, "billion": [3, 12, 21, 31], "bin": [7, 25], "binari": [0, 3, 5, 7, 9, 10, 12, 28], "binarycrossentropi": 4, "bind": 0, "binomi": [21, 25, 28], "binsboot": [6, 32], "bioinformat": 0, "biolog": [1, 12], "bios1100": [21, 28], "bird": [0, 3], "birth": 28, "bishop": [27, 28], "bit": [1, 4, 19, 22, 25, 28], "bitwis": 25, "bivari": 2, "bk": [13, 31], "bla": [22, 28], "black": [8, 9, 14], "blame": [], "block": [6, 10, 21, 22, 25, 28, 32], "blockquot": [], "blog": 28, "blogpost": 4, "blue": [0, 3], "bm": [], "bmatrix": [0, 1, 3, 5, 7, 8, 11, 13, 22, 28, 29, 30, 31], "bmi": 1, "bodi": [0, 1, 4, 12], "bold": 1, "boldfac": [0, 5, 16, 29, 30], "boldsymbol": [0, 1, 2, 3, 5, 6, 7, 8, 10, 11, 13, 14, 16, 17, 19, 23, 28, 30, 31], "boltzmann": [12, 21, 28], "book": [17, 23, 27, 28, 29, 32], "book1": 27, "bool": [], "boolean": [4, 17], "boost": [1, 9, 21, 28], "boostrap": 10, "bootstrap": [1, 13, 19, 21, 23, 28, 31], "born": 31, "borrow": 28, "boston_dataset": [], "bot": 8, "both": [0, 1, 4, 5, 6, 8, 9, 10, 13, 14, 15, 16, 17, 19, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "bottl": 7, "bottou": 31, "bound": [8, 12, 31], "boundari": [2, 4, 8, 11, 12], "bousquet": 31, "bower": [], "box": [4, 9], "boyd": [8, 13, 30], "bracket": [4, 25], "brain": [1, 7, 12], "branch": [9, 28], "break": [0, 4, 6, 11, 14, 28, 31], "breast": [5, 7, 11], "breviti": 13, "brew": [0, 21, 23, 28], "brg": 8, "brian": [], "brief": [23, 29], "briefli": [0, 16, 19, 28, 32], "bring": [0, 5, 6, 10, 29, 31], "britt": [26, 28], "broad": 0, "broadli": 28, "brought": [13, 21, 28], "brownle": 4, "browser": [15, 28], "brute": [3, 5, 11, 29], "bsd": [], "budget": 31, "buffer_s": 4, "bug": [], "bugfix": [], "bui": 4, "build": [0, 4, 5, 6, 10, 16, 22, 25, 28, 32], "built": [1, 3, 4, 6, 32], "bunch": 11, "bundl": [], "busi": [], "byte": [22, 28], "c": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 19, 20, 21, 22, 24, 25, 26, 27, 29, 30, 31, 32], "c1": [8, 11], "c2": [8, 11], "c4a2f5": [], "c5e478": [], "c9d1d9": [], "c_": [8, 9, 10, 13, 25, 30, 31], "c_0": 25, "c_1": 12, "c_2": 12, "c_3": 12, "c_4": 12, "c_i": [12, 13, 31], "c_k": 25, "ca": [1, 28], "caab6d": [], "cach": 10, "cal": [0, 8, 10, 12, 13, 30, 31], "calcul": [0, 1, 2, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 16, 19, 22, 25, 28, 31, 32], "california": 23, "call": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 18, 19, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "calor": [0, 29], "caltech": [], "cambridg": [13, 27, 30], "can": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27, 29, 30], "cancel": [0, 13, 28, 29], "cancer": [5, 10], "cancerpd": 7, "candid": [8, 9, 10, 31], "cannot": [0, 1, 4, 5, 6, 7, 8, 9, 23, 25, 29, 30, 31], "canopi": [0, 21, 23, 28], "canva": [15, 16, 19, 20, 23, 28], "cap": 5, "capabl": [0, 1, 8, 13, 21, 28], "capac": [2, 26], "capita": [], "caption": [20, 23], "captur": [4, 11, 12, 28], "car": [3, 4], "card": [0, 7, 28], "cardin": 1, "care": [11, 15, 19, 31], "carefulli": [13, 31], "carlo": [0, 6, 21, 25, 27, 28, 32], "carri": [2, 6, 7, 23, 32], "cart": 10, "case": [0, 1, 2, 3, 4, 5, 6, 7, 11, 12, 13, 14, 15, 16, 21, 22, 23, 28, 32], "casella": 27, "cast": 1, "cat": [3, 4], "catch": 0, "categor": [0, 1, 3, 9, 11, 28], "categori": [0, 1, 3, 7, 10, 12, 14, 28], "categorical_crossentropi": [1, 3], "caus": [0, 5, 6, 25, 28, 29, 30, 31, 32], "causal": 0, "causat": [0, 28], "cax": 1, "cb": [6, 28], "cbar": 1, "cc": [0, 1, 5, 13, 28, 29, 30, 31], "cc398b": [], "ccbb44": [], "ccc": [5, 12, 30], "cdf": 25, "cdot": [0, 2, 6, 12, 13, 14, 22, 25, 28, 30, 31, 32], "celebr": [13, 30], "cell": 4, "center": [0, 1, 6, 7, 8, 9, 11, 14, 18, 23, 25, 28, 29, 31, 32], "central": [0, 3, 5, 6, 8, 16, 20, 22, 28, 29], "centroid": [14, 25], "centroid_differ": 14, "centuri": 3, "certain": [0, 3, 6, 7, 9, 25, 28, 29, 32], "certainti": 32, "cf": [], "cf222e": [], "cffi": [], "cg": 13, "cha": [], "chain": [0, 1, 13, 21, 25, 28], "challeng": 15, "chanc": [1, 5, 13, 25, 31], "chang": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 13, 14, 15, 16, 19, 22, 23, 25, 28, 29, 30, 31, 32], "changelog": [], "channel": 3, "chapter": [0, 6, 10, 11, 16, 17, 19, 22, 23, 27, 28, 29, 30, 31, 32], "chapter3": [0, 23], "charact": [0, 3, 5, 28, 29, 30], "character": [8, 9, 10, 12, 25], "characterist": [0, 1, 3, 10, 13, 28], "charg": [0, 28], "charl": [], "charset": [], "chase": 4, "chatgpt": [15, 23], "chd": 7, "chddata": 7, "cheap": [5, 29, 30, 31], "cheaper": [1, 13, 31], "check": [1, 3, 4, 5, 11, 13, 15, 16, 19, 22, 28, 31], "checkmark": 3, "checkpoint": 4, "checkpoint_dir": 4, "checkpoint_prefix": 4, "chen": 10, "cheng": 29, "chiaramont": 2, "childcar": 16, "children": 16, "choic": [0, 1, 2, 3, 4, 6, 9, 12, 13, 14, 20, 22, 28, 29, 30, 31, 32], "choleski": [5, 22, 29, 30], "choos": [2, 3, 6, 9, 10, 11, 13, 14, 15, 18, 19, 23, 30, 32], "chosen": [0, 1, 2, 6, 8, 9, 10, 13, 16, 25, 28, 30, 31, 32], "chosen_datapoint": 1, "christian": 27, "christoph": [27, 28], "chunk": 31, "cifar": 3, "cifar10": 3, "circ": [1, 12, 31], "circl": [0, 8, 12, 29, 31], "circuit": 3, "circumfer": 9, "circumv": [1, 5, 13, 29, 30, 31], "citat": [], "cite": [20, 23], "ckpt": 4, "claim": [], "clariti": 25, "class": [0, 1, 3, 4, 6, 7, 8, 9, 11, 12, 13, 25, 28, 32], "class_nam": [3, 9], "class_val": 9, "class_valu": 9, "classic": [7, 9, 13], "classif": [0, 3, 5, 6, 7, 8, 11, 12, 21, 23, 27, 28, 29, 32], "classifi": [0, 1, 4, 7, 9, 10, 11, 28], "classificaton": 1, "classifii": 10, "claus": [], "clean": 1, "clear": [1, 5, 10, 12, 13, 31], "clearli": [0, 3, 5, 6, 7, 8, 25, 29, 30, 32], "clever": [1, 10], "clf": [0, 6, 8, 9, 10, 28, 29], "clf3": 0, "clf_lasso": 6, "clf_ridg": 6, "cli": 15, "click": [], "clip": [3, 25, 31], "clock": 31, "clone": [15, 26], "close": [0, 1, 2, 4, 6, 8, 9, 11, 12, 13, 14, 18, 25, 27, 28, 30, 31, 32], "closer": [3, 5, 13, 29, 30, 31], "closest": [8, 11, 13, 14], "closur": [21, 28], "cloud": [21, 28], "cluster": [0, 1, 4, 6, 11, 21, 28, 32], "cluster_label": 14, "cm": [1, 2, 3, 6, 8, 13, 30, 31], "cmap": [0, 1, 2, 3, 4, 6, 8, 9, 10, 28], "cmap_arg": 6, "cmd": [9, 15], "cn_": 25, "cnn": 12, "cnn_kera": 3, "cntk": [21, 28], "co": [0, 2, 3, 6, 9, 13, 28, 32], "code": [0, 3, 4, 6, 7, 8, 18, 19, 21, 22, 25, 27], "codec": [], "coef": [0, 28], "coef0": 8, "coef_": [0, 5, 6, 8, 9, 13, 16, 28, 29, 30, 31], "coeff": 5, "coeffici": [0, 3, 5, 6, 7, 8, 9, 13, 18, 22, 28, 29, 31, 32], "coerc": [0, 6, 28, 32], "coin": [10, 25], "coin_toss": 10, "col": [0, 11, 28, 29], "colab": [21, 28], "cold": 9, "colinear": [], "collabor": [20, 23], "collaps": 8, "collect": [2, 6, 10, 11, 17, 21, 25, 27, 28, 32], "collinear": [5, 29, 30], "color": [0, 3, 4, 6, 8, 9, 10, 25, 31], "color_channel": 3, "color_cod": 6, "colorbar": [1, 6, 20], "colsample_bytre": 10, "colsaobject": 10, "column": [0, 1, 2, 5, 6, 7, 8, 9, 11, 12, 16, 17, 18, 19, 22, 28, 29, 30, 31, 32], "columntransform": 9, "com": [4, 6, 15, 16, 19, 20, 21, 23, 27, 28, 30, 31, 32], "combin": [1, 2, 5, 6, 7, 10, 15, 18, 25, 32], "come": [0, 1, 3, 4, 5, 12, 13, 14, 15, 28, 29, 30, 31], "comfort": [], "command": [0, 1, 15], "comment": [0, 4, 5, 6, 20, 23], "commerci": [0, 21, 23, 28], "commit": 15, "commod": [0, 28], "common": [0, 1, 3, 5, 6, 7, 9, 11, 13, 14, 16, 23, 25, 28, 29, 30, 31, 32], "commonli": [0, 1, 4, 6, 7, 9, 13, 14, 29, 31, 32], "commonmark": [], "commun": [0, 12, 15, 23], "commut": 3, "commutatitav": 3, "compact": [0, 1, 3, 5, 6, 7, 9, 11, 12, 13, 14, 28, 29, 32], "compair": 0, "compar": [0, 3, 4, 5, 6, 11, 13, 18, 22, 23, 28, 29, 30, 31, 32], "comparison": [2, 4, 13], "compat": 7, "compens": 31, "compet": 0, "competit": 10, "compil": [0, 1, 3, 4, 13, 21, 22, 28], "complet": [0, 2, 3, 4, 9, 12, 15, 16, 17, 18, 19, 20, 28], "completenn": 12, "complex": [1, 5, 8, 9, 11, 12, 13, 16, 19, 28, 30, 31, 32], "complianc": [], "complic": [0, 1, 9, 13, 23, 28, 30, 31, 32], "compoment": 29, "compon": [0, 1, 3, 4, 5, 6, 7, 9, 14, 16, 21, 28, 29, 30, 32], "components_": 11, "compos": [9, 12, 13, 14, 21, 28], "compphys": [0, 6, 16, 20, 21, 23, 24, 26, 27, 28, 29, 30], "compress": [0, 28, 29], "compris": 6, "compromis": [5, 29, 30], "compulsori": [21, 28], "comput": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 15, 16, 17, 18, 21, 22, 23, 24, 25, 27, 28, 29, 30, 32], "computation": [0, 3, 6, 9, 13, 25, 28, 30, 31], "computationalscienceuio": 28, "computerlab": 23, "concaten": [2, 4, 6, 14], "concav": [1, 13, 29, 30], "concentr": 10, "concept": [0, 2, 21, 28, 29], "conceptu": [12, 13, 30], "concern": [0, 1, 4, 7, 28, 30], "concic": 28, "conclud": [0, 5, 13, 31], "conclus": 1, "cond": 2, "conda": [0, 1, 21, 23, 28], "condis": 29, "condit": [0, 2, 4, 5, 6, 8, 9, 11, 13, 25, 28, 29, 31, 32], "conduct": 21, "condwav": 2, "confid": [0, 5, 6, 7, 8, 19, 28, 29], "configur": 3, "confirm": [5, 12], "conform": [], "confus": [5, 6, 7, 10, 22, 29, 32], "confusion_matrix": 9, "congruenti": 25, "conjug": [4, 8], "conjugaci": 13, "conjunct": 3, "connect": [0, 1, 3, 4, 9, 11, 12, 13, 22, 28, 29, 30], "consensu": 31, "consequ": [5, 6, 8, 10, 12, 13, 29, 30, 31, 32], "consequenti": [], "conserv": [5, 14, 29, 30], "consid": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 16, 19, 22, 23, 25, 28, 29, 30, 31, 32], "consider": [0, 1, 5, 13, 28, 29, 30, 32], "consist": [1, 2, 3, 4, 6, 12, 13, 23, 25, 29, 30, 32], "consol": [], "const": [], "constant": [0, 2, 4, 5, 6, 8, 12, 13, 16, 18, 25, 28, 29, 30, 31], "constitu": [0, 28], "constitut": [2, 6, 32], "constrain": [1, 3, 5, 7, 11, 30], "constraint": [5, 6, 8, 13, 29, 30, 32], "construct": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 22, 25, 28, 29, 32], "constructor": [], "consum": 31, "contact": [0, 28], "contain": [0, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 18, 19, 22, 23, 25, 27, 28, 29, 30, 31, 32], "contemporari": 28, "content": [1, 15, 20, 21, 22, 28, 30, 31], "context": [6, 10, 13, 23, 30, 31, 32], "contigu": 22, "contin": 19, "continu": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 19, 22, 23, 25, 28, 29, 30, 31, 32], "contour": [9, 10, 13], "contourf": [8, 9, 10], "contract": [], "contrast": [1, 4, 9, 10, 12, 28, 31], "contribut": [0, 3, 5, 13, 18, 25, 28, 29, 30, 31], "contributor": [0, 23], "control": [0, 1, 3, 9, 13, 15, 21, 28], "conv": [3, 4], "conv2d": [3, 4], "conv2dtranspos": 4, "convei": 28, "conveni": [5, 6, 12, 13, 22, 23, 28, 30, 31, 32], "convent": [12, 29], "converg": [1, 2, 4, 5, 8, 13, 14, 18, 29, 30], "convergencewarn": [], "convers": [20, 31], "convert": [0, 1, 4, 5, 9, 11, 13, 22, 28, 29, 30], "converttomatrix": 4, "convex": [4, 5, 7, 29], "convinc": [13, 30], "convolut": [1, 4, 21, 28], "cool": [4, 9], "coolwarm": 6, "coordin": [5, 12, 14, 29, 30, 31], "coorel": [], "copi": [0, 1, 14, 15, 29], "copyright": [], "core": 10, "corel": 28, "coronari": 7, "corr": [5, 7, 11, 29], "correalt": [11, 21], "correct": [0, 1, 2, 3, 4, 5, 7, 13, 15, 19, 20, 22, 25, 28, 29, 30, 32], "correctli": [1, 2, 6, 7, 10, 18, 19, 23, 32], "correl": [0, 1, 3, 5, 6, 7, 10, 12, 13, 21, 25, 28, 30, 31, 32], "correlation_matrix": [5, 7, 11, 29], "correspond": [0, 3, 5, 6, 8, 9, 11, 12, 21, 22, 23, 25, 28, 29, 30, 32], "cortex": 12, "cosin": [3, 6, 32], "cost": [0, 2, 3, 5, 6, 7, 8, 9, 12, 13, 16, 17, 18, 19, 23, 28], "cost_deep_grad": 2, "cost_funct": 2, "cost_function_deep": 2, "cost_function_deep_grad": 2, "cost_function_grad": 2, "cost_grad": 2, "cost_histori": [], "cost_ol": [], "cost_ridg": [], "cost_sum": 2, "costli": 31, "costol": [13, 31], "could": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 17, 18, 22, 23, 25, 28, 29, 30, 31, 32], "coulomb": [0, 28], "count": [0, 9, 15, 24, 25, 26, 28], "counteract": 31, "counterpart": 28, "countor": 13, "coupl": [4, 5, 6, 32], "cours": [0, 1, 3, 5, 11, 15, 16, 17, 19, 20, 23, 26, 29, 32], "coursework": 15, "courvil": [27, 28, 29, 31], "cov": [5, 6, 11, 22, 25, 28, 29, 32], "cov_xi": [5, 11, 29], "cov_xx": [5, 11, 29], "cov_yi": [5, 11, 29], "covari": [0, 7, 21, 22, 28, 30], "covariance_matrix": [5, 11, 14], "cover": [0, 5, 21, 26, 27, 29, 30, 32], "covert": [0, 28], "covxi": 25, "covxx": 25, "covxz": 25, "covyi": 25, "covyz": 25, "covzz": 25, "cpu": 1, "craft": 3, "crash": 31, "creat": [1, 3, 4, 5, 9, 10, 11, 12, 15, 18, 19, 21, 28, 31], "create_biases_and_weight": 1, "create_convolutional_neural_network_kera": 3, "create_neural_network_kera": 1, "create_x": [5, 11], "creation": [], "credit": [0, 7, 26, 28], "crim": [], "crime": [], "criteria": [0, 4, 9, 10, 14, 25, 28], "criterion": [9, 10, 13, 18, 30, 31], "critic": [6, 23, 29], "critiqu": 23, "cross": [0, 1, 3, 7, 9, 10, 13, 15, 21, 25, 28, 29, 30, 31], "cross_entropi": 4, "cross_val_scor": [6, 32], "cross_valid": [7, 10], "crossvalid": [6, 32], "crucial": [1, 25, 31], "cs231": 3, "csr_matrix": [22, 28], "css": [], "csv": [0, 4, 6, 7, 9, 32], "ctnk": 1, "cubic": 0, "culprit": [], "cumbersom": [5, 32], "cumprod": [], "cumsum": [10, 11, 28], "cumul": [7, 10, 25, 31], "cumulative_heads_ratio": 10, "cup": 5, "current": [1, 2, 3, 4, 13, 14, 15, 16, 27, 30, 31], "curs": [0, 29], "curv": [6, 7, 10, 12, 23], "curvatur": [13, 30, 31], "custom": [6, 14], "custom_cmap": [9, 10], "custom_cmap2": [9, 10], "custom_lin": [], "cutpoint": 9, "cv": [6, 7, 10, 32], "cvxbook": [13, 30], "cvxopt": [5, 8, 29], "cycl": [1, 12], "cycler": [], "d": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 17, 19, 20, 22, 25, 26, 28, 29, 30, 31, 32], "d1": [], "d166a3": [], "d2": [], "d2_g_t": 2, "d2a8ff": [], "d4d0ab": [], "d71835": [], "d9dee3": [], "d_f": [13, 30], "d_g_t": 2, "d_net_out": 2, "da": 3, "dagger": [5, 22, 29, 30], "dai": [1, 9, 21], "damag": [], "damp": 3, "darget": 9, "darkr": 25, "dat": [0, 28], "dat_id": [0, 6, 7, 9, 28, 32], "data": [2, 4, 5, 8, 10, 12, 13, 14, 16, 19, 20, 22, 23, 27, 30, 31, 32], "data1": 14, "data2": 14, "data3": 14, "data4": 14, "data_id": [0, 6, 7, 9, 28, 32], "data_indic": 1, "data_panda": 28, "data_path": [0, 6, 7, 9, 28, 32], "databas": 1, "datafil": [0, 6, 7, 9, 28, 32], "datafram": [0, 4, 5, 7, 9, 11, 28, 29], "datapoint": [1, 5, 6, 7, 11, 13, 16, 30, 31, 32], "datasci": [15, 16, 19], "dataset": [0, 4, 6, 7, 8, 9, 10, 11, 13, 14, 16, 23, 28, 30, 31, 32], "datatyp": 4, "date": [15, 18, 23, 28, 29, 30, 31, 32], "daughter": 10, "davi": [], "david": 27, "davison": 32, "dbb7ff": [], "dbh": 1, "dbo": 1, "dcc6e0": [], "dcomposit": 22, "ddot": 2, "de": 31, "dead": 1, "deadlin": [15, 20], "deal": [0, 1, 3, 5, 6, 8, 11, 13, 14, 19, 22, 25, 28, 29, 30, 31], "dealt": 0, "debt": 7, "debug": [0, 5, 6, 29, 30, 31, 32], "debugg": [], "decad": [0, 3, 31], "decai": [0, 13, 25, 28], "decemb": [26, 28], "decent": 10, "decid": [0, 2, 3, 5, 6, 9, 18, 29, 30, 31, 32], "decim": [0, 28], "decis": [0, 1, 8, 11, 21, 27, 28], "decision_funct": 8, "decision_tre": 9, "decisiontreeclassifi": [9, 10], "decisiontreeregressor": [0, 9, 10], "declar": [0, 4, 20, 22, 28], "declare_namespac": [], "decompos": [5, 6, 22, 29, 30], "decomposit": [0, 6, 12, 28], "decompost": [5, 29, 30], "deconvolut": 3, "decorrel": [10, 13, 31], "decreas": [1, 2, 4, 5, 6, 10, 11, 13, 19, 30, 31, 32], "dedic": 20, "deduc": [0, 28], "deep": [3, 7, 12, 13, 21, 27, 29, 30], "deep_neural_network": 2, "deep_param": 2, "deep_tree_clf": [9, 10], "deep_tree_clf1": 9, "deep_tree_clf2": 9, "deepen": [5, 21, 28], "deeper": [0, 3, 4, 28], "deeplearningbook": [27, 28, 30, 31], "deer": 3, "def": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 16, 17, 25, 28, 29, 30, 31, 32], "def_covari": 25, "default": [0, 1, 2, 4, 6, 7, 22, 28, 29], "default_tim": 4, "defect": [5, 29, 30], "defici": [5, 29, 30], "defin": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 18, 19, 22, 23, 25, 29, 30, 31, 32], "definit": [1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 22, 25, 29, 30, 31, 32], "defint": 25, "degre": [3, 5, 6, 8, 9, 10, 11, 15, 16, 19, 20, 23, 25, 28, 30, 31, 32], "deisenroth": 29, "del": 1, "delet": [6, 15], "delimit": 4, "deliv": [15, 24, 28], "delta": [0, 2, 3, 6, 8, 12, 13, 14, 28, 31], "delta_": [1, 22], "delta_0": 3, "delta_1": 3, "delta_2": 3, "delta_3": 3, "delta_4": 3, "delta_5": 3, "delta_h": [0, 1, 28], "delta_j": [3, 12], "delta_k": 12, "delta_l": [1, 3], "delta_momentum": [13, 31], "delta_n": [0, 3, 28], "delug": 21, "delv": 0, "demand": [13, 30], "demonstr": [0, 3, 5, 6, 7, 11, 12, 19, 21, 28, 29, 30, 31, 32], "den": 4, "denomin": [1, 5, 31], "denot": [1, 2, 6, 7, 13, 25, 30, 31], "dens": [1, 3, 4], "densiti": [0, 2, 6, 25, 32], "depart": [26, 28, 29, 30, 31, 32], "depend": [0, 1, 2, 4, 5, 6, 7, 8, 11, 12, 13, 15, 16, 21, 22, 23, 25, 28, 29, 30, 31], "depict": 25, "deploy": [0, 21, 23, 28], "depth": [0, 3, 9, 10, 22, 32], "der": [], "deriv": [0, 1, 2, 6, 7, 8, 10, 11, 13, 18, 21, 23, 28], "derivati": 13, "derivative_fn": 13, "descend": [5, 9, 11, 29, 30], "descent": [0, 1, 3, 7, 8, 12, 28, 29], "describ": [0, 2, 4, 5, 6, 8, 10, 11, 12, 13, 19, 20, 22, 23, 28, 31, 32], "descript": [0, 8, 9, 20, 23, 28], "design": [0, 1, 3, 4, 5, 6, 7, 10, 11, 12, 13, 17, 18, 23, 28, 30, 31, 32], "designmatrix": [0, 28], "desir": [0, 2, 4, 5, 13, 14, 28, 29, 30, 31], "desktop": 15, "despit": [1, 12, 31], "destroi": 22, "det": [5, 22, 29, 30], "detail": [0, 6, 11, 13, 14, 18, 22, 23, 29, 30, 31], "detect": [3, 8, 12], "determin": [0, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 18, 22, 25, 28, 29, 30, 31, 32], "determinist": [7, 13, 25, 30, 31], "dev": [1, 23], "develop": [0, 3, 5, 8, 10, 11, 12, 21, 22, 23, 28, 29], "deviat": [0, 1, 2, 4, 5, 6, 17, 18, 19, 23, 25, 28, 29, 31, 32], "devis": 12, "df": [4, 8, 11, 13, 28], "df1": 28, "di": [], "diag": [5, 8, 29, 30, 31], "diagnost": [1, 10], "diagon": [0, 5, 7, 13, 18, 19, 22, 25, 28, 29, 30, 31], "diagonaliz": [5, 29, 30], "diagram": 10, "diagsvd": 6, "dice": [6, 25, 32], "dict": [6, 8], "dictionari": [], "did": [0, 1, 5, 6, 7, 10, 11, 14, 16, 23, 28, 32], "die": 1, "diff": 2, "diff1": 2, "diff2": 2, "diff_ag": 2, "diffeent": 8, "differ": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 32], "differenti": [0, 3, 16, 21, 22, 28, 29, 30], "difficult": [0, 1, 6, 10, 13, 25, 28, 31, 32], "difficulti": [0, 1, 13, 28, 30, 31], "diffonedim": 2, "digit": [0, 1, 3, 4, 6, 26, 28], "dilemma": [13, 31], "dilut": 1, "dim": [4, 11, 14, 22], "dimens": [0, 1, 2, 3, 4, 5, 8, 11, 14, 16, 22, 28, 29, 30], "dimension": [0, 4, 5, 6, 9, 11, 13, 14, 19, 21, 22, 23, 28, 29, 30, 31, 32], "dimensionless": [0, 3, 28], "diment": 22, "diminish": 31, "dimnsion": 4, "diod": 3, "direct": [0, 1, 2, 4, 11, 12, 13, 14, 28, 29, 30, 31], "directli": [1, 4, 5, 6, 18, 25, 29, 30], "directori": [], "disadvantag": [0, 28, 31], "disappear": [3, 6, 32], "disc_loss": 4, "disc_tap": 4, "discard": [6, 11, 31, 32], "disciplin": [0, 3, 12], "disclaim": 25, "discord": 28, "discourag": [13, 15, 30], "discov": [0, 28], "discover": 5, "discret": [1, 3, 5, 7, 13], "discrimin": [4, 7, 10, 11], "discriminator_loss": 4, "discriminator_loss_list": 4, "discriminator_model": 4, "discriminator_optim": 4, "discuss": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 19, 20, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "diseas": 7, "disguis": [6, 29, 31], "disk": 31, "disord": [1, 7], "displai": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 23, 25, 28, 29, 31, 32], "displaystyl": [0, 5, 17, 28, 29, 30, 31], "disregard": [0, 28], "dissimilar": [11, 14], "dist": 14, "distanc": [8, 9, 11, 14, 25], "distance_list": 9, "distinct": [3, 7, 8, 9, 10, 14], "distinctli": 8, "distinguish": [0, 4, 7, 8, 25, 28], "distplot": [], "distribut": [0, 1, 4, 6, 7, 10, 11, 13, 14, 18, 19, 21, 22, 23, 28, 29, 30, 31], "distrubut": [0, 21, 23, 28], "div": [], "dive": [0, 8, 22, 28], "diverg": [1, 13, 30, 31], "divid": [0, 1, 3, 5, 6, 7, 8, 9, 11, 12, 18, 19, 25, 28, 29, 31, 32], "divis": [6, 8, 9, 13, 18, 22, 25, 31, 32], "dl": [], "dm": [], "dna": 7, "dnn": [0, 1, 2, 4, 12, 28], "dnn1": 4, "dnn2_gru2": 4, "dnn_kera": 1, "dnn_model": 1, "dnn_numpi": 1, "dnn_scikit": [0, 1, 28], "do": [0, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 22, 23, 28, 29, 30, 32], "doc": [0, 15, 16, 19, 21, 23, 24, 26, 27, 28], "document": [4, 13, 15], "docutil": [], "doe": [0, 1, 2, 3, 4, 5, 6, 8, 10, 11, 12, 13, 15, 16, 17, 18, 19, 22, 23, 25, 28, 31, 32], "doesn": [3, 9, 12, 28, 31], "dog": [1, 3, 4], "dollar": [], "domain": [5, 8, 13, 23, 30, 32], "domcontentload": [], "domin": [0, 28], "don": [0, 1, 3, 5, 6, 8, 11, 13, 15, 16, 21, 23, 28, 29, 31], "done": [0, 2, 3, 4, 5, 6, 9, 10, 11, 13, 16, 20, 22, 23, 28, 29, 30, 31, 32], "dot": [0, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 18, 22, 23, 25, 28, 29, 30, 31, 32], "doubl": [3, 4, 16, 22, 28], "doubli": 1, "doubt": 23, "down": [0, 3, 6, 9, 11, 12, 13, 30, 31], "download": [0, 1, 3, 5, 6, 15, 20, 22, 27, 28], "downsampl": 3, "dozen": 1, "dq": [6, 32], "draft": 20, "drag": 13, "dragon": [], "dramat": 11, "drastic": 4, "draw": [4, 6, 10, 13, 30, 32], "drawback": [0, 1, 3, 13, 29, 30, 31], "drawn": [1, 4, 6, 7, 11, 25, 28, 32], "drive": [3, 4], "driven": 3, "drop": [0, 1, 5, 6, 11, 13, 25, 28, 29, 30, 32], "dropna": [0, 6, 28, 32], "dropout": 4, "dt": [2, 3, 13, 25], "dtype": [0, 1, 3, 4, 14, 22, 28], "dual": [], "dub": [0, 28], "duboi": [], "due": [1, 2, 5, 6, 8, 10, 12, 13, 18, 26, 28, 29, 30, 31, 32], "dugard": [], "dummi": [], "dure": [0, 1, 3, 4, 8, 9, 11, 20, 21, 23, 28, 31, 32], "dwell": [], "dwh": 1, "dwo": 1, "dx": [2, 3, 8, 25], "dx_1": 25, "dx_1p": [6, 32], "dx_2p": [6, 32], "dx_mp": [6, 32], "dx_n": 25, "dxp": [6, 32], "dy": [1, 8, 25], "dynam": 4, "dz": 8, "e": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 25, 26, 28, 29, 30, 31, 32], "e1e1e1": [], "e_": [0, 2, 28], "each": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 21, 22, 24, 25, 26, 28, 29, 30, 31, 32], "eager": 32, "eapprox": [0, 28], "earli": [1, 13, 31], "earlier": [0, 5, 7, 8, 9, 11, 12, 13, 20, 28, 29], "earthexplor": 6, "eas": [6, 9, 14, 32], "easi": [0, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 21, 22, 28, 29, 30, 31, 32], "easier": [5, 6, 8, 9, 13, 15, 20, 23, 25, 28, 29, 30, 32], "easiest": [13, 18], "easili": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 22, 23, 28, 29, 30, 31, 32], "eastern": [26, 28], "ebind": [0, 28], "eblock": 9, "ec8e2c": [], "econom": [], "econometr": 28, "economi": 5, "ecosystem": [21, 28], "ect": 24, "edg": 3, "edgecolor": [6, 32], "edit": [], "editor": [15, 20], "edu": [13, 23, 30], "educ": [0, 23, 28, 32], "ee6677": [], "eff": 25, "effect": [1, 4, 10, 13, 16, 17, 18, 25, 31], "effic": 1, "effici": [0, 3, 10, 13, 21, 22, 25, 28, 31], "effort": 19, "efron": [6, 32], "egrad": 13, "eig": [5, 11, 13, 22, 25, 28, 29, 30, 31], "eigen": 25, "eigenpair": [5, 11, 29, 30], "eigenvalu": [0, 5, 8, 11, 13, 22, 28, 29, 30, 31], "eigenvector": [5, 11, 13, 29, 30], "eight": [22, 28], "eigval": [22, 25, 28], "eigvalu": [11, 13, 30, 31], "eigvec": [22, 25, 28], "eigvector": [11, 13, 30, 31], "eir": [26, 28], "eispack": [22, 28], "either": [1, 5, 6, 7, 8, 9, 10, 11, 13, 18, 19, 23, 25, 28, 29, 30, 31, 32], "eivind": 26, "eivinsto": 26, "ekstr\u00f8m": 4, "elabor": 25, "elarn": 3, "electr": [0, 3, 12, 28], "electron": 28, "eleg": 11, "element": [1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 19, 20, 21, 22, 23, 27, 29, 31, 32], "elementari": [10, 13, 22], "elementwis": [3, 13], "elementwise_grad": [2, 13], "elessar": 28, "elif": 14, "elim": 22, "elimin": [3, 8], "elin": [26, 28], "ell_": [], "ellipsi": 16, "els": [1, 3, 4, 7, 9, 12, 13, 16, 22], "elu": 1, "elus": [0, 28], "em": [], "email": [20, 24, 26, 28], "emb": [], "embed": [0, 11, 29], "embodi": [6, 23, 32], "emit": 25, "emner": 27, "emph": 31, "emphas": [0, 10, 21, 28], "emphasi": [0, 21, 27, 28], "empir": [1, 11, 25], "emploi": [0, 1, 5, 6, 11, 13, 25, 28, 29, 30, 32], "employ": 0, "empti": [6, 10, 15, 32], "emul": 12, "en": [21, 23, 27], "enabl": [11, 31], "enbodi": [6, 32], "encod": [0, 3, 5, 9, 11, 14, 28, 29, 30], "encompass": [0, 23, 25], "encount": [0, 1, 5, 7, 13, 15, 23, 25, 28, 29, 30, 31], "encourag": [15, 23], "end": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 20, 22, 25, 26, 28, 29, 30, 31, 32], "endblock": [], "endfor": [], "endif": [], "endors": [], "endpoint": [3, 6], "energi": [0, 4, 6, 32], "enforc": 12, "eng": 27, "engin": [0, 1, 3, 4, 21, 28], "english": 23, "enjoi": 31, "enocurag": 23, "enorm": 3, "enough": [0, 6, 13, 28, 30, 31, 32], "ensembl": [1, 9, 28], "ensur": [0, 1, 2, 3, 5, 6, 11, 13, 18, 25, 29, 30, 31, 32], "entail": 28, "enter": [5, 6, 29, 30, 31], "enthought": [0, 21, 23, 28], "entir": [1, 3, 7, 9, 21, 25, 28, 31], "entireti": [], "entiti": [9, 12, 22, 28], "entri": [0, 5, 8, 11, 12, 22, 28, 29, 31, 32], "entropi": [1, 3, 7, 10, 13, 28, 30, 31], "enumer": [0, 1, 2, 3, 4, 6, 8, 28, 29, 31], "env": 25, "environ": [2, 21, 23, 28], "environemnt": 15, "eo": [0, 6, 32], "eol": 0, "eosfit": 0, "epoch": [0, 1, 3, 4, 12, 13, 28, 31], "eppstein": [], "epsilon": [0, 5, 6, 7, 13, 23, 28, 29, 30, 31, 32], "epsilon_": [0, 28], "epsilon_0": [0, 28], "epsilon_1": [0, 28], "epsilon_2": [0, 28], "epsilon_i": [0, 28, 29], "eq": [3, 13, 14, 22, 25, 30], "eqnarrai": [3, 5, 6, 32], "equal": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 13, 14, 16, 18, 22, 23, 25, 28, 29, 30, 31, 32], "equat": [1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 17, 19, 22, 25, 28, 31, 32], "equilibrium": [2, 12], "equiv": [3, 13, 22, 25, 30, 31], "equival": [0, 1, 5, 7, 8, 11, 13, 21, 22, 28, 29, 30, 31, 32], "equivel": 19, "eras": [], "erf": 25, "eriador": 28, "eric": [], "err": [0, 10], "err_": [6, 32], "err_sqr": 2, "errat": [13, 30, 31], "erron": 2, "error": [1, 2, 4, 5, 6, 7, 9, 11, 12, 13, 15, 16, 17, 18, 19, 21, 22, 23, 25, 31], "error_estimate_corr_tim": 25, "error_hidden": 1, "error_output": 1, "escap": [13, 30, 31], "escapehtml": [], "especi": [1, 3, 9, 12, 13, 15, 18, 23, 31], "essenti": [0, 5, 6, 9, 10, 12, 14, 15, 23, 25, 29, 30, 31], "establish": [0, 6, 10, 11, 16, 23], "estim": [0, 1, 5, 6, 7, 10, 11, 13, 19, 21, 25, 28, 29, 30, 31], "estimated_mse_fold": [6, 32], "estimated_mse_kfold": [6, 32], "estimated_mse_sklearn": [6, 32], "et": [0, 2, 4, 16, 17, 20, 27, 28, 29, 30, 32], "eta": [0, 1, 3, 8, 12, 13, 18, 28, 30, 31], "eta0": [8, 13], "eta_": 13, "eta_j": 31, "eta_t": [13, 31], "eta_v": [0, 1, 3, 28], "etc": [0, 1, 3, 5, 7, 8, 9, 11, 12, 13, 14, 21, 22, 23, 25, 29, 30, 31], "ethic": 21, "etsim": 32, "euclidean": [0, 14, 29, 31], "euler": [], "evalu": [0, 2, 3, 4, 5, 6, 9, 13, 15, 16, 17, 19, 23, 25, 28, 29, 30, 31, 32], "evalut": [13, 23], "even": [0, 1, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 21, 22, 25, 28, 29, 30, 31, 32], "evenli": 4, "event": [5, 7, 10, 25, 32], "eventu": [0, 5, 6, 11, 12, 13, 23, 26, 29, 30, 31, 32], "everi": [0, 1, 2, 3, 4, 5, 6, 9, 10, 11, 12, 13, 14, 15, 21, 25, 26, 28, 29, 30, 31, 32], "everyth": [4, 12, 16, 18], "everywher": [4, 13, 30], "evolv": 0, "exact": [0, 5, 11, 12, 13, 22, 25, 28, 29, 31], "exactli": [0, 3, 4, 6, 12, 18, 21, 29, 31, 32], "exam": 28, "examin": [6, 32], "exampl": [0, 5, 11, 12, 13, 15, 16, 18, 20, 21, 22, 23, 25, 27], "exce": [1, 12, 13, 31], "exceed": 31, "excel": [0, 1, 4, 5, 10, 20, 23, 28, 29], "except": [3, 4, 6, 8, 9, 22], "excess": [0, 28], "exchang": 31, "excit": 0, "exclud": [1, 6, 12, 29, 31, 32], "exclus": [0, 1, 3, 6, 25, 28, 32], "execut": [2, 5, 13, 15, 29, 30, 31], "exemplari": [], "exemplifi": [13, 31], "exercic": [26, 28], "exercis": [5, 21, 23, 24, 26, 28, 30, 31, 32], "exhaust": [6, 31, 32], "exhibit": [0, 5, 6, 8, 28, 29, 32], "exist": [0, 1, 2, 3, 5, 6, 7, 8, 9, 13, 19, 22, 23, 28, 30, 31, 32], "exit": [5, 22, 29, 30], "exp": [0, 1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 16, 17, 19, 25, 29, 30, 31, 32], "exp_term": 1, "expand": [5, 7, 11, 13, 30], "expans": [0, 3, 5, 8, 10, 12, 13, 28, 29, 30], "expect": [0, 1, 5, 6, 7, 11, 12, 13, 15, 18, 21, 23, 28, 29, 31], "expectation_value_of_h_wrt_p": 25, "expens": [6, 10, 13, 16, 30, 31], "experi": [0, 1, 6, 8, 13, 15, 21, 23, 28, 29, 30, 31, 32], "experiment": [0, 4, 6, 9, 25, 28, 32], "expert": [1, 9], "explain": [0, 6, 9, 10, 11, 13, 16, 19, 23, 28, 30], "explained_variance_ratio_": 11, "explan": [], "explanatori": [0, 28], "explicit": [0, 3, 6, 13, 22, 23, 28, 29, 30, 31], "explicitli": [0, 4], "explod": 1, "exploit": [0, 3, 12, 13, 28, 31], "explor": [1, 4, 6, 8, 13, 18, 21, 23, 28, 30, 31], "expon": 1, "exponenti": [0, 1, 5, 6, 10, 13, 25, 28, 30], "export": [9, 15, 16, 19, 20], "export_graphviz": 9, "export_text": 9, "exporttext": 9, "expos": 21, "express": [0, 2, 3, 5, 6, 7, 10, 12, 13, 18, 22, 23, 25, 28, 30, 31, 32], "exptmean": 25, "exptvari": 25, "extend": [0, 2, 7, 11, 13, 21, 28, 31], "extend_path": [], "extens": [0, 12, 15, 21, 28], "extent": [0, 1, 6, 27, 32], "extern": [3, 6, 9], "extra": [1, 3, 5, 15, 26, 28, 29, 30], "extract": [0, 3, 5, 6, 7, 8, 11, 13, 16, 17, 22, 28, 29], "extrapol": [0, 28], "extrem": [0, 1, 4, 5, 6, 7, 8, 9, 13, 15, 16, 22, 29, 30, 31], "extremum": [13, 30], "extrins": 11, "ey": [0, 5, 6, 13, 14, 18, 22, 28, 29, 30, 31], "f": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 12, 13, 14, 15, 16, 17, 18, 19, 22, 25, 26, 28, 29, 30, 31, 32], "f1": 13, "f11": [0, 28], "f12": [0, 28], "f13": [0, 28], "f1_grad": 13, "f1d": 13, "f2": 13, "f26196": [], "f2_grad_x1": 13, "f2_grad_x1_analyt": 13, "f2_grad_x2": 13, "f2_grad_x2_analyt": 13, "f2f2f2": [], "f3": 13, "f3_grad": 13, "f3_grad_analyt": 13, "f4": 13, "f4_grad": 13, "f4_grad_analyt": 13, "f5": 13, "f5_grad": 13, "f5a394": [], "f5ab35": [], "f5f5f5": [], "f6": 13, "f6_for": 13, "f6_for_grad": 13, "f6_grad_analyt": 13, "f6_while": 13, "f6_while_grad": 13, "f7": 13, "f78c6c": [], "f7_grad": 13, "f7_grad_analyt": 13, "f8": 13, "f8_grad": 13, "f8f8f2": [], "f9": [0, 13, 28], "f9_altern": 13, "f9_alternative_grad": 13, "f9_grad": 13, "f_": 10, "f_0": [3, 10], "f_1": [10, 13, 30], "f_2": [12, 13, 30], "f_3": 12, "f_d": 25, "f_grad": 13, "f_grad_analyt": 13, "f_i": [0, 6, 12, 16, 32], "f_m": [3, 10], "f_n": 3, "f_vec": 2, "face": [13, 28, 30], "facecolor": [6, 8, 25, 32], "facil": [0, 21], "facilit": 12, "fact": [0, 1, 3, 5, 9, 11, 12, 13, 28, 29, 30, 31], "facto": 31, "factor": [0, 1, 3, 5, 6, 9, 10, 11, 13, 22, 25, 28, 29, 30], "factori": 13, "fad000": [], "fade": 6, "fae4c2": [], "fafab0": [9, 10], "fail": [0, 6, 13, 26, 28, 30, 32], "failur": 7, "fairli": [1, 2, 18, 25, 31], "faisal": [16, 29], "fake": 4, "fake_loss": 4, "fake_output": 4, "fall": [8, 9, 24], "fals": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 14, 16, 17, 22, 28, 29, 30, 31, 32], "famili": [0, 7, 8, 25, 29, 31], "familiar": [0, 3, 5, 6, 8, 15, 21, 22, 23, 25, 28, 32], "famou": [6, 12], "far": [0, 3, 4, 5, 6, 8, 11, 12, 13, 14, 16, 20, 28, 29, 30, 31], "fashion": [0, 9, 10, 28, 31], "fast": [1, 3, 6, 10, 12, 13, 21, 25, 28, 30, 31, 32], "faster": [1, 11, 13, 31], "fastest": [13, 22, 30], "fatal": [], "favor": [7, 31], "favorit": 25, "fc": 3, "fcfcfc": [], "fdac54": [], "fdf2e2": [], "featur": [0, 1, 3, 5, 6, 7, 8, 10, 11, 12, 13, 15, 17, 18, 19, 21, 25, 28, 30, 31, 32], "feature_nam": [1, 7, 9], "feautur": 9, "fed": 1, "feed": [0, 2, 3, 11, 21, 28], "feed_forward": 1, "feed_forward_out": 1, "feed_forward_train": 1, "feedback": [4, 20, 28], "feeddorward": 4, "feedforward": [1, 4, 12], "feel": [0, 5, 6, 11, 13, 15, 16, 18, 21, 23, 26, 28], "feet": [], "fefef": [], "fefeff": [], "felt": 23, "fenc": [], "fernando": [], "fetch": [6, 15], "few": [1, 3, 4, 5, 9, 17, 18, 19, 25, 28], "fewer": [0, 9, 11, 19, 28, 31], "ff7b72": [], "ff9492": [], "ffa07a": [], "ffa657": [], "ffb757": [], "ffd700": [], "ffd900": [], "ffd9002e": [], "ffffff": [], "ffnn": [1, 12], "fi": [], "field": [0, 3, 6, 12, 19, 21], "fieldmask": [], "fifth": [0, 6, 28], "fig": [0, 1, 2, 3, 4, 6, 7, 12, 13, 14, 23, 28], "fig_id": [0, 6, 7, 9, 28, 32], "figaxi": 25, "figsiz": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 28, 32], "figur": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 16, 21, 23, 28, 29, 30, 31, 32], "figure_id": [0, 6, 7, 9, 28, 32], "figurefil": [0, 6, 7, 9, 28, 32], "file": [0, 4, 5, 6, 7, 9, 15, 20, 23, 28, 32], "file_prefix": 4, "filenam": 28, "fill": [5, 9, 18, 29, 30], "fill_valu": [], "filter": [3, 4], "final": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 18, 20, 23, 24, 25, 26, 28, 30, 32], "financ": 0, "find": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 21, 23, 25, 28, 29, 30, 31], "fine": [0, 14], "finish": [2, 20], "finit": [3, 5, 6, 12, 13, 17, 25, 29, 30, 32], "finnicki": 15, "fire": [], "first": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 18, 19, 22, 23, 25, 26, 27, 29, 31, 32], "first_moment": 31, "first_term": 31, "firsteigvector": 11, "fit": [1, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 17, 18, 19, 23, 25, 29, 31, 32], "fit_beta": 29, "fit_intercept": [0, 5, 6, 16, 29, 30, 31, 32], "fit_mod": 9, "fit_theta": [6, 31], "fit_transform": [0, 6, 8, 9, 11, 15, 19, 32], "fiti": [0, 28], "five": [0, 9, 28, 29], "fix": [0, 3, 4, 6, 10, 11, 12, 13, 23, 28, 32], "flag": 4, "flat": [12, 13, 30, 31], "flatten": [1, 3, 4, 5, 22], "flavor": [], "flexibl": [1, 6, 8, 10, 12, 28, 31, 32], "flip": [26, 28], "float": [0, 3, 4, 5, 9, 11, 13, 14, 22, 28, 29, 30], "float32": [4, 9], "float64": [4, 22, 28], "flop": [5, 22, 29, 30], "flow": [1, 4, 12], "fluctuat": [5, 31], "fly": 11, "fm": 0, "fmax": 3, "fmesh": 13, "fn": 7, "focu": [0, 3, 4, 5, 6, 15, 21, 23, 27, 28, 29, 30, 31, 32], "focus": [1, 6, 7, 22, 29, 31], "fold": [6, 9, 23], "folder": [0, 4, 6, 15, 20, 23, 28], "follow": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 19, 20, 21, 22, 23, 25, 26, 27, 28, 29, 30, 31, 32], "font": [7, 20, 25, 28], "fontdict": 25, "fontsiz": [1, 6, 8, 9, 10, 25], "fontweight": 1, "footprint": [3, 31], "foral": [8, 29], "forc": [0, 5, 6, 10, 11, 29, 30, 31], "forcast": 4, "forcier": [], "forecast": [4, 12], "forest": [0, 1, 9, 21, 28], "forget": [11, 31], "form": [0, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 16, 21, 22, 23, 25, 28, 29, 30, 31, 32], "formal": [3, 4, 14, 18, 25], "format": [0, 1, 3, 4, 6, 7, 8, 9, 10, 11, 20, 21, 25, 27, 32], "format_data": 4, "formatstrformatt": [6, 13, 30, 31], "formatt": [], "formul": [4, 6, 11, 14], "formula": [3, 13, 25, 30], "forth": [4, 12], "fortran": [0, 21, 22, 28], "fortran2003": [21, 28], "fortran2008": 23, "fortran90": 25, "fortun": [0, 11, 29], "forward": [0, 3, 6, 21, 22, 28, 31, 32], "found": [1, 2, 4, 5, 6, 12, 13, 19, 20, 23, 28, 29, 31, 32], "foundat": [21, 28], "four": [4, 5, 6, 8, 12, 22, 24, 26, 28, 30], "fourier": [0, 28], "fourierdef1": 3, "fourierdef2": 3, "fourierseriessign": 3, "fourth": [12, 28, 29], "fp": 7, "frac": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "fraction": 9, "frame": [7, 31], "framework": [1, 8, 10, 25], "frank": [5, 11], "frankefunct": [5, 6, 11], "fredli": [26, 28], "free": [0, 6, 11, 13, 15, 16, 18, 21, 22, 23, 25, 26, 27, 28], "freecodecamp": 21, "freedom": [5, 30], "freeli": [0, 23], "freez": 15, "frequenc": [3, 6, 7, 25, 32], "frequent": [0, 8, 9, 13, 30], "frequentist": 21, "fresh": 10, "fridai": [15, 26, 28], "friedman": [6, 19, 23, 27, 28], "friendli": 4, "fro": 23, "frodo": 28, "frog": 3, "from": [0, 1, 2, 3, 4, 6, 7, 8, 9, 11, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27], "from_cod": 9, "from_logit": [3, 4], "from_tensor_slic": 4, "front": [0, 4, 5, 28, 29, 30], "frustrat": 15, "fulfil": [2, 5, 12, 29, 30], "full": [1, 3, 5, 7, 9, 10, 13, 25, 28, 29, 30], "full_matric": [5, 29, 30], "fulli": [3, 6, 12, 25, 32], "fullnam": [], "fun": [21, 28], "func": 2, "function": [2, 3, 4, 5, 9, 14, 15, 16, 17, 18, 19, 20, 21, 22], "functionali": 11, "fundament": [0, 6, 21, 28, 32], "funtion": 2, "furnish": [], "further": [2, 7, 9, 19, 28], "furthermor": [0, 3, 5, 6, 7, 11, 12, 13, 21, 23, 28, 29, 30, 31, 32], "futur": [0, 4, 8, 9, 28], "fy": [15, 23, 24, 26, 27, 28], "fys4155": 23, "fys5419": [27, 28], "fys5429": [27, 28], "f\u00f8470": [26, 28], "g": [0, 1, 2, 3, 4, 6, 8, 9, 10, 11, 13, 15, 18, 19, 25, 28, 29, 30, 31, 32], "g0": 2, "g_": [2, 9, 10, 31], "g_0": 2, "g_1": [2, 10], "g_2": [2, 10], "g_analyt": 2, "g_dnn_ag": 2, "g_euler": 2, "g_i": 2, "g_m": [3, 10], "g_n": 3, "g_re": 2, "g_t": [2, 31], "g_t_d2t": 2, "g_t_d2x": 2, "g_t_dt": 2, "g_t_hessian": 2, "g_t_hessian_func": 2, "g_t_jacobian": 2, "g_t_jacobian_func": 2, "g_trial": 2, "g_trial_deep": 2, "g_vec": 2, "gain": [1, 5, 7, 9, 10, 13, 29, 30], "galleri": [0, 28], "game": 4, "gamge": 28, "gamma": [0, 2, 8, 9, 10, 11, 13, 28, 30], "gamma1": 8, "gamma2": 8, "gamma_": [0, 28], "gamma_0": 10, "gamma_1": 10, "gamma_1x": 10, "gamma_i": [0, 8, 25, 28], "gamma_j": 13, "gamma_k": [13, 30], "gamma_m": 10, "gamma_x": [0, 28], "gap": [8, 31], "gate": [4, 12], "gather": [0, 1, 12, 29], "gaug": 12, "gaussbacksub": 22, "gaussian": [4, 5, 6, 8, 14, 18, 25, 28, 32], "gaussian_point": 14, "gaussian_rbf": 8, "gave": [13, 31], "gavra": 28, "gbc": 28, "gca": [2, 6, 8, 13], "gd": [1, 30], "gd_clf": 10, "gdclassiffiercgain": 10, "gdclassiffierconfus": 10, "gdclassiffierroc": 10, "gdm": 13, "gdregress": 10, "ge": [1, 5, 7, 25, 29, 30], "gen_loss": 4, "gen_tap": 4, "gender": [0, 28], "genener": 4, "gener": [0, 1, 2, 3, 5, 6, 8, 10, 11, 12, 13, 14, 15, 16, 18, 20, 22, 23, 25, 27, 29, 30, 31, 32], "generaliz": 16, "generallay": 12, "generate_and_save_imag": 4, "generate_imag": 4, "generate_latent_point": 4, "generate_simple_clustering_dataset": 14, "generated_imag": 4, "generator_loss": 4, "generator_loss_list": 4, "generator_model": 4, "generator_optim": 4, "genom": 21, "geodes": 11, "geoff": 31, "geometr": [0, 13, 28, 31], "geometri": 5, "georg": 27, "geotif": 6, "geq": [2, 5, 8, 9, 13, 29, 30, 31], "gerard": [], "geron": [0, 27, 28], "get": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 13, 15, 19, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "get_dummi": 9, "get_paramet": 2, "get_split": 9, "get_yaxi": 8, "get_yticklabel": 6, "getmask": [], "gh": 15, "giant": 31, "gibb": [21, 28], "gif": 4, "gini": 10, "gini_index": 9, "ginvers": 13, "git": [0, 15, 21, 28], "gitcdn": [], "giter": [13, 31], "github": [0, 20, 21, 23, 24, 26, 27, 28, 29], "gitignor": 15, "gitlab": [0, 15, 21, 23, 28], "give": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 14, 18, 19, 21, 23, 25, 28, 29, 30, 31, 32], "given": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 19, 22, 25, 28, 29, 30, 31, 32], "global": [6, 7, 13, 30, 31], "glorot": 1, "gmail": [], "gnew": 13, "go": [0, 1, 3, 5, 6, 8, 9, 11, 12, 13, 15, 16, 18, 28, 29, 30, 32], "goal": [0, 7, 9, 28], "goe": [0, 1, 2, 5, 6, 13, 14, 15, 19, 22, 28, 29, 30, 31, 32], "goessner": [], "golden": 13, "gone": [5, 29, 30], "gong": 1, "good": [1, 3, 4, 5, 6, 9, 10, 11, 13, 15, 18, 21, 25, 27, 29, 30, 31], "goodfellow": [4, 27, 28, 29, 30], "googl": [1, 4, 21, 28], "got": [1, 6, 23], "gotten": 28, "gov": 6, "govern": 28, "gp": 27, "gpu": [1, 13, 21, 28, 31], "grad": [2, 13, 31], "grad_analyt": 13, "grad_ol": 18, "grad_ridg": 18, "grade": [23, 24], "gradient": [0, 3, 4, 7, 8, 9, 12, 21, 28, 29], "gradient_desc": 31, "gradientboostingclassifi": 10, "gradientboostingregressor": 10, "gradients_of_discrimin": 4, "gradients_of_gener": 4, "gradienttap": 4, "gradual": [1, 14], "grai": [4, 6], "granger": [], "grant": [], "graph": [1, 9, 11, 12, 13, 16, 20, 30, 31], "graph_from_dot_data": 9, "graphic": [0, 1, 9, 15, 28], "grasp": 0, "gray_r": [1, 3], "grayscal": 3, "great": [5, 13, 15, 30, 31], "greater": [1, 7, 25, 29], "greatli": 13, "greedi": 9, "green": [0, 3, 9, 25], "grei": 4, "grid": [1, 3, 6, 7, 8, 12, 25, 29, 31, 32], "grossli": [13, 30], "ground": [0, 28], "group": [0, 6, 7, 9, 14, 15, 20, 21, 23, 24, 26, 28, 32], "groupbi": [0, 28], "grow": [1, 3, 9, 10, 31], "growth": [0, 28], "gru": 4, "guarante": [0, 4, 13, 25, 28, 29, 30, 31], "guess": [1, 4, 10, 13, 14, 30, 31], "guestrin": 10, "gui": 15, "guid": 1, "guidelin": 20, "g\u00f6ssner": [], "h": [0, 1, 5, 6, 8, 13, 15, 19, 25, 26, 27, 28, 29, 30, 31], "h1": 2, "h_": [0, 13, 28, 30, 31], "h_0": 31, "h_1": [2, 13, 30], "h_2": [2, 13, 30], "h_m": 10, "h_t": 31, "ha": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "haanen": [26, 28], "habit": [0, 29], "had": [0, 1, 6, 7, 13, 28, 30, 31, 32], "hadamard": [1, 12, 13, 31], "half": [1, 8, 9], "halv": 10, "hand": [0, 1, 2, 3, 5, 11, 12, 13, 21, 22, 23, 25, 26, 27, 28, 29, 30, 31], "handi": [3, 23], "handl": [0, 1, 2, 5, 9, 11, 15, 18, 21, 29, 30, 31], "handle_unknown": 9, "handsid": 12, "handwrit": 12, "handwritten": [1, 5], "happen": [1, 2, 3, 4, 5, 6, 10, 13, 25, 29, 30, 31], "hard": [1, 7, 8, 10, 13, 30, 31], "hardcopi": [21, 28], "harder": [0, 1, 19, 29], "harmon": 3, "hash": 31, "hasn": [], "hassl": [0, 21, 28], "hast": [21, 28], "hasti": [0, 6, 16, 17, 19, 20, 23, 27, 28, 29, 32], "hat": [0, 1, 5, 6, 7, 9, 10, 11, 12, 13, 16, 17, 18, 19, 22, 29, 30, 31, 32], "hauser": [], "have": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "have_sys_un_h": [], "haven": 1, "he": 7, "head": [4, 10, 25], "header": [0, 28], "heads_proba": 10, "health": [0, 29], "hear": [0, 13, 28, 31], "heart": [0, 7, 28], "heatmap": [0, 1, 3, 7, 17, 20, 28], "heavi": 31, "heavili": 0, "heavisid": 1, "height": [1, 3, 6, 29], "held": [13, 31], "help": [0, 1, 4, 12, 13, 15, 16, 23, 28, 31, 32], "helper": [4, 14], "henc": [0, 5, 6, 8, 9, 10, 12, 13, 28, 29, 30, 31, 32], "henrik": [26, 28], "her": 7, "here": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "hereaft": [0, 8, 12, 28], "herebi": [], "hermitian": 22, "hessenberg": 22, "hessian": [0, 2, 5, 13], "heterogen": [9, 10], "hex": [], "hi": 7, "hidden": [1, 3, 4, 12], "hidden_bia": 1, "hidden_bias_gradi": 1, "hidden_layer_s": [0, 1, 28], "hidden_neuron": 4, "hidden_weight": 1, "hidden_weights_gradi": 1, "hierarch": [5, 29, 30], "high": [0, 1, 2, 3, 4, 5, 6, 9, 10, 11, 13, 14, 21, 22, 23, 28, 29, 30, 31, 32], "higher": [0, 1, 3, 5, 6, 8, 13, 18, 23, 28, 29, 30, 31, 32], "highest": [1, 2], "highli": [0, 3, 4, 10, 19, 21, 22, 27, 28, 29, 30, 31], "highlight": [], "highwai": [], "hing": 8, "hint": [13, 15, 16, 29, 30], "hinton": 31, "hip": 21, "hire": 0, "hist": [4, 6, 7, 25, 32], "histogram": [6, 7, 25], "histor": [7, 11], "histori": [3, 4, 12, 15, 31], "hitherto": 5, "hjorth": [26, 28, 29, 30, 31, 32], "hobbi": 25, "hoc": [5, 29, 30], "hoff": 27, "hold": [1, 3, 6, 13, 14, 30, 31, 32], "holder": [0, 28], "holdgraf_evidence_2014": [], "home": [], "homepag": [23, 28], "homework": [6, 13, 30, 31], "homogen": [1, 3, 9, 10, 13, 31], "honchar": 2, "hopefulli": [0, 11, 15, 19, 25, 28, 31], "horizont": 11, "horlyk": [26, 28], "hors": [3, 7, 28], "hot": [1, 9], "hour": [1, 21, 24, 25, 26, 28, 31, 32], "how": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "howev": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 25, 28, 29, 30, 31, 32], "href": [], "hspace": [0, 4, 8, 10, 25, 28], "hstack": 1, "htf": 28, "html": [0, 16, 20, 21, 23, 24, 26, 27, 28, 29, 30, 31], "http": [0, 3, 4, 6, 13, 15, 16, 19, 20, 21, 22, 23, 24, 26, 27, 28, 29, 30, 31, 32], "huang": [0, 28], "huber": [0, 28], "huge": [1, 3, 4, 21, 31], "human": [0, 1, 3, 6, 9, 12, 29], "humid": 9, "hundr": 1, "hungri": 1, "hybrid": 24, "hydrogen": [0, 28], "hyperbol": [1, 4, 12], "hyperparam": 8, "hyperparamet": [3, 4, 5, 6, 9, 13, 18, 23, 29, 30, 31], "hyperplan": 11, "h\u00f8rlyk": [26, 28], "i": [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 29, 30, 31, 32], "i0": [0, 28], "i1": [0, 6, 8, 12, 28, 29, 31], "i2": [0, 8, 12, 28], "i3": [0, 12, 28], "i5": [0, 28], "i_": [13, 30, 31], "i_1": [5, 6, 32], "i_2": [5, 6, 32], "i_t": 31, "ian": 27, "ic": [1, 23], "id": [7, 13, 30, 31], "ida": [26, 28], "idea": [0, 1, 2, 3, 4, 6, 9, 10, 12, 13, 20, 22, 23, 29, 30, 31, 32], "ideal": [0, 2, 6, 8, 13, 25, 28, 31, 32], "idem": [6, 32], "ident": [5, 6, 12, 13, 17, 18, 22, 29, 30], "identical": 32, "identifi": [0, 1, 7, 9, 11, 12, 13, 14, 28, 29], "idx": [], "ieor": 25, "ifi": 27, "ifs": [21, 28], "ignor": [0, 1, 3, 9, 15, 29, 31], "ii": [22, 25], "iii": [22, 28], "ij": [0, 1, 3, 6, 8, 12, 14, 16, 22, 25, 28, 29, 31], "ik": [0, 22, 28, 29], "iki": [], "ill": 31, "illinoi": [], "illustr": [5, 7, 10, 12, 13, 14, 20, 21, 28], "ilsvrc": 31, "im": 6, "imag": [1, 3, 4, 6, 9, 11, 12, 14, 27, 28], "image_at_epoch_": 4, "image_batch": 4, "image_height": 3, "image_path": [0, 6, 7, 9, 28, 32], "image_width": 3, "imageio": 6, "imagenet": 31, "images_from_seed_imag": 4, "imagin": 1, "immedi": [0, 3, 4, 6, 21, 28, 31], "implement": [0, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 19, 20, 23, 25, 28, 29, 30, 31], "impli": [3, 5, 6, 7, 13, 22, 29, 30, 31, 32], "implicit": [3, 31], "implicitli": [11, 25], "import": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 23, 25, 31, 32], "importantli": 3, "importerror": [], "impos": [0, 6, 11, 12, 28], "imposs": [0, 5, 28, 29, 30], "impract": 31, "impress": [0, 12, 28], "improv": [0, 4, 5, 9, 10, 11, 13, 15, 23, 29, 30], "impur": 9, "imread": 6, "imshow": [1, 3, 4, 6], "in3050": [27, 28], "in3310": 28, "in4080": [27, 28], "in4300": [27, 28], "in4310": 27, "in5400": 3, "in5550": 27, "in_out_neuron": 4, "inaccur": [13, 30], "inact": 12, "inadequ": [0, 28], "inappropri": 31, "inch": [6, 29], "incident": [], "includ": [0, 1, 2, 3, 4, 5, 6, 7, 11, 12, 15, 16, 17, 18, 19, 20, 21, 25, 26, 27, 28, 29, 30, 32], "include_bia": [6, 9, 32], "incom": [12, 16], "incorrect": 1, "incoveni": 8, "increas": [0, 1, 3, 4, 5, 6, 9, 12, 13, 19, 23, 25, 28, 29, 31, 32], "increasingli": 25, "increment": 31, "ind": 6, "inde": [0, 2, 4, 5, 6, 13, 28, 29, 30], "indefinit": 4, "independ": [0, 5, 6, 7, 8, 12, 13, 25, 28, 29, 30, 31], "index": [0, 1, 3, 4, 10, 14, 21, 22, 23, 25, 27, 28], "index_col": [0, 28], "indic": [0, 1, 3, 4, 5, 6, 9, 10, 11, 13, 16, 23, 28, 29], "indirect": [], "indispens": [6, 32], "individu": [1, 6, 7, 10, 12, 25, 28, 29, 31, 32], "indu": [], "indx": 22, "indx1": 2, "indx2": 2, "indx3": 2, "ineffici": [3, 13], "inequ": [8, 13], "inequaltii": 30, "inertia": 13, "inexperi": [], "inf": [], "inf1000": [21, 28], "inf1100": [21, 28], "inf1100l": [21, 28], "inf1110": [21, 28], "inf3000": 28, "infeas": [9, 31], "infer": [0, 1, 4, 6, 27, 28, 32], "inferenc": 1, "infil": [0, 6, 7, 9, 28, 32], "infin": [5, 6, 7, 11, 19, 29, 30, 32], "infinit": [3, 31], "infinitesim": 25, "influenc": [6, 10, 18, 32], "influenti": 1, "info": 28, "inform": [0, 1, 3, 4, 6, 9, 11, 12, 13, 14, 22, 23, 27, 28, 30, 31, 32], "inforom": 15, "infrequ": 31, "infti": [3, 6, 13, 25, 30, 32], "ingeni": [13, 30, 31], "ingredi": [0, 9, 28], "inher": [6, 31, 32], "inherit": [22, 28, 31], "init": [], "initi": [0, 1, 2, 6, 10, 13, 14, 18, 22, 25, 28, 30, 31, 32], "inject": 14, "inlin": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 25, 28, 29, 30, 31, 32], "inner": [0, 13, 29], "innerhtml": [], "inp": 4, "inplac": 13, "input": [0, 1, 3, 4, 5, 6, 7, 8, 12, 13, 14, 16, 23, 25, 28, 29, 30, 31, 32], "input_dim": 1, "input_shap": [3, 4], "inputs": 1, "inputs_shuffl": [0, 1, 29], "inquiri": 20, "insert": [3, 5, 6, 8, 10, 25, 29, 30, 32], "insid": [4, 7], "insight": [0, 1, 5, 21, 28, 29, 30, 32], "insist": [6, 13, 29, 31], "inspir": [0, 1, 12, 23, 28], "instabl": 2, "instal": [0, 1, 5, 6, 9, 15, 20], "instanc": [0, 1, 2, 4, 6, 9, 11, 13, 16, 28, 29, 30, 31, 32], "instanti": 10, "instead": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 13, 14, 17, 20, 22, 25, 28, 29, 31, 32], "institut": 1, "instruct": [0, 1, 15], "int": [0, 1, 2, 3, 4, 5, 6, 11, 13, 14, 22, 25, 29, 31, 32], "int32": 10, "int_": [3, 6, 25, 32], "int_0": 25, "int_a": 25, "intak": [0, 29], "integ": [1, 2, 13, 14, 22, 25, 28], "integer_vector": 1, "integr": [3, 6, 25, 28, 32], "intellig": [0, 14, 27, 28], "intend": 10, "intens": [1, 18], "intention": 14, "interact": [0, 6, 9, 12, 21, 23, 28], "intercept": [0, 6, 8, 11, 13, 16, 17, 18, 19, 28, 29, 30, 31, 32], "intercept_": [0, 6, 8, 9, 13, 28, 29, 31], "interchang": [5, 12, 22], "interconnect": 1, "interesit": [], "interest": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 12, 19, 21, 23, 25, 28, 29, 30, 32], "interfac": [0, 1, 15, 22, 29], "interior": [0, 9, 28], "intermedi": [22, 29, 31], "intern": [1, 10, 12], "internation": [], "interpol": [1, 3, 4, 6, 12], "interpr": [5, 29, 30], "interpret": [0, 1, 6, 9, 10, 12, 13, 15, 16, 22, 23, 25], "interrupt": [], "interv": [0, 3, 5, 6, 7, 13, 19, 25, 28, 29, 30], "intial": [13, 30], "intract": [0, 4, 29], "intrins": [3, 11, 22, 25, 28], "intro": [21, 27, 28], "introduc": [0, 1, 5, 6, 8, 10, 12, 22, 23, 25, 28, 30, 31, 32], "introduct": [1, 2, 4, 13, 27, 29, 30, 31], "introductori": [0, 4, 22, 27, 28, 29], "intuit": [0, 5, 6, 8, 12, 13, 23, 28, 31, 32], "inv": [0, 5, 13, 17, 28, 29, 30, 31], "invalid": [], "invalu": [0, 13, 21, 28, 30], "invari": 1, "invd": 5, "inver": 8, "invers": [0, 3, 6, 13, 28, 29, 30, 31], "inverse_transform": 8, "invert": [0, 5, 7, 10, 13, 16, 18, 28, 31], "investig": [], "invh": [13, 31], "invok": 8, "involv": [0, 2, 6, 7, 11, 12, 28, 29, 31, 32], "io": [0, 21, 23, 24, 26, 27, 28, 29], "ion": [], "ip": [0, 8, 25, 28], "ipca": 11, "ipynb": [21, 28], "ipython": [0, 5, 7, 9, 11, 14, 21, 23, 28, 29], "iq": [6, 32], "iri": [8, 9], "irreduc": [6, 32], "irrelev": [5, 29, 30], "irrespect": [0, 28], "irvin": 23, "isaac": [], "isaacmus": [], "iseffici": [], "isn": 5, "isnul": [], "isomap": 11, "issu": [1, 9, 15, 22, 31], "it_arrai": 13, "item": [0, 13, 28], "items": [22, 28], "iter": [1, 2, 4, 6, 8, 13, 14, 18, 23, 25, 30, 31, 32], "its": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 20, 21, 22, 23, 25, 28, 30, 31, 32], "itself": [5, 6, 12, 23, 25, 28, 29, 32], "j": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 13, 14, 15, 16, 22, 23, 25, 27, 28, 29, 30, 31, 32], "j1": 22, "j_": 6, "j_41hld6ttu": 32, "j_lasso_sk": 6, "j_ridge_sk": 6, "j_sk": 6, "jackknif": [6, 21, 28, 32], "jacobian": [2, 13, 30], "janko": [], "jason": 4, "javascript": [], "jax": [21, 28, 31], "jeff": [], "jensen": [26, 28, 29, 30, 31, 32], "jerom": [19, 23, 27], "jhauser": [], "ji": [12, 22], "jit": 13, "jj": [0, 5, 6, 28, 32], "jk": [0, 1, 6, 12, 22, 28], "jl": [0, 28], "jm": 22, "jnp": 13, "job": [2, 8, 10, 15], "join": [0, 4, 6, 7, 9, 28, 32], "joint": [4, 5], "jonathan": [], "json": [], "judg": [13, 30], "judgement": 6, "julia": [21, 22, 23], "jump": [25, 31], "junk": 4, "jupit": 28, "jupyt": [0, 15, 16, 19, 21, 23, 27, 28, 32], "jupyterbook": [], "jupytext": [], "just": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 20, 21, 25, 28, 29, 30, 31, 32], "justif": 0, "justifi": [3, 10], "k": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 21, 22, 23, 25, 26, 28, 29, 30, 31], "k0": 7, "k1": 7, "kaggl": [6, 23], "kappa_d": 25, "karl": [26, 28], "karush": 8, "katex": [], "katrin": [26, 28], "keep": [0, 1, 4, 5, 6, 11, 13, 14, 15, 18, 22, 23, 28, 29, 30, 31, 32], "keepdim": [1, 6, 10, 22, 32], "kei": [1, 3, 6, 12, 31], "kellei": [], "kenneth": [], "kept": [4, 6, 14, 32], "kera": [0, 4, 21, 23, 28], "kernel": [0, 1, 3, 21, 28, 29], "kernel_regular": [1, 3], "kernel_s": 4, "kernelpca": 11, "kev": [0, 28], "kevin": [27, 28], "kevinsheppard": [], "keyword": [18, 22, 28], "kfold": [6, 32], "kg": 1, "ki": 22, "kick": [1, 13, 31], "kiener": 2, "kilomet": [6, 29], "kim": [], "kind": [0, 2, 3, 4, 8, 12, 13, 14, 28, 29], "kingma": 31, "kj": [6, 12, 22, 29, 31], "kjm": [21, 28], "kkt": 8, "kl": 25, "km": [12, 28], "kmean": 14, "kmeanspoint": 14, "kn_k": 14, "know": [0, 1, 2, 5, 6, 8, 13, 15, 16, 17, 19, 20, 21, 28, 29, 30], "knowledg": [0, 21, 28], "known": [1, 3, 4, 5, 6, 7, 8, 9, 12, 18, 22, 23, 25, 27, 29, 31, 32], "kondev": [0, 28], "kp": 25, "kpca": 11, "kramdown": [], "kroneck": 14, "kt": [], "kuhn": 8, "kvalsund": [26, 28], "kwown": [0, 28], "l": [0, 1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 13, 22, 23, 25, 28, 30, 31], "l0": 7, "l1": [0, 1, 3, 7, 28], "l1_l2": [1, 3], "l1regl": 5, "l2": [1, 3], "l_": [22, 31], "l_1": 7, "l_2": [7, 13, 30, 31], "l_i": 31, "l_j": 12, "la": 13, "la_": [], "la_i": 12, "la_k": 12, "lab": [20, 21, 23, 28], "label": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 15, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "labelencod": [7, 10], "labels": [6, 8, 9], "labels_shuffl": [0, 1, 29], "laboratori": 24, "lack": [0, 28, 31], "lagari": 2, "lagrang": [8, 11], "lam": 18, "lambda": [0, 1, 2, 3, 5, 6, 7, 8, 10, 12, 13, 17, 18, 19, 20, 23, 25, 28, 29, 30, 31, 32], "lambda_": 11, "lambda_0": 11, "lambda_1": [5, 8, 11, 29, 30], "lambda_2": [8, 11], "lambda_i": [8, 11], "lambda_iy_i": 8, "lambda_jy_iy_j": 8, "lambda_k": 8, "lambda_n": [5, 8, 29, 30], "lamda": 1, "land": 8, "landmark": 8, "landscap": [13, 18, 30, 31], "langl": [0, 6, 11, 25, 28, 29], "languag": [0, 1, 4, 8, 21, 22, 23, 27, 28], "lapack": [22, 28], "laplac": 5, "laptop": [15, 21], "larg": [0, 1, 2, 4, 5, 6, 8, 9, 10, 11, 13, 18, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "larger": [0, 3, 5, 6, 8, 10, 11, 13, 17, 25, 28, 29, 30, 31, 32], "largest": [4, 8, 11], "lasso": [0, 7, 21, 28, 31, 32], "lasso_sk": 6, "last": [0, 1, 3, 4, 5, 6, 7, 8, 12, 16, 17, 19, 22, 23, 25, 26, 28, 30, 32], "latent": 4, "latent_dim": 4, "latent_point": 4, "latent_space_value_rang": 4, "later": [0, 1, 4, 7, 8, 12, 13, 14, 15, 19, 21, 23, 28, 31], "latest": [4, 15, 21], "latest_checkpoint": 4, "latex": [20, 28], "latexcodec": [], "latter": [0, 3, 6, 7, 8, 11, 13, 22, 25, 28, 29, 30, 31, 32], "lattic": 12, "law": 0, "layer": [0, 4, 13, 28, 31], "lbfg": [7, 9, 10], "lc_messag": [], "lcc": [5, 6, 32], "lda": 11, "ldot": [0, 6, 11, 23, 28, 32], "le": [5, 7, 10, 13, 17, 25, 29, 30, 31], "lead": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 22, 25, 28, 29, 30, 31, 32], "leaf": 9, "leaki": 1, "leakyrelu": 4, "lear": [13, 30], "learn": [3, 4, 5, 6, 7, 8, 9, 10, 12, 22, 26, 27], "learnabl": 3, "learner": 10, "learnig": 28, "learning_r": [8, 10], "learning_rate_init": [0, 1, 28], "learning_schedul": [13, 31], "learnt": 23, "least": [0, 7, 8, 10, 11, 17, 18, 21, 22, 25, 32], "leat": [13, 31], "leav": [0, 1, 3, 5, 6, 9, 11, 28, 30, 32], "lectur": [0, 1, 5, 10, 11, 12, 13, 21, 22, 23, 24, 26, 27, 29], "lecturenot": [0, 21, 23, 27, 28], "left": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "leftarrow": [8, 12], "legend": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 15, 28, 29, 30, 31, 32], "leinonen": 28, "len": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 16, 17, 22, 28, 29, 30, 31, 32], "length": [0, 1, 3, 4, 8, 9, 13, 16, 21, 28, 29, 30, 31], "length_of_sequ": 4, "leq": [0, 5, 7, 8, 13, 14, 25, 28, 29, 30, 31], "less": [0, 1, 3, 4, 5, 6, 8, 9, 13, 21, 25, 28, 29, 30, 31, 32], "lessen": 1, "let": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 19, 22, 25, 28, 29, 30, 31, 32], "letter": [0, 16, 22, 25, 28, 29], "level": [0, 1, 5, 6, 9, 21, 22, 23, 24, 26, 28, 31, 32], "leverag": 31, "lexer": [], "li": [8, 11], "liabil": [], "liabl": [], "lib": [], "liberti": 31, "liblinear": 10, "librari": [0, 1, 2, 3, 4, 5, 6, 9, 10, 11, 22, 23, 25, 27, 29, 30, 31], "licenc": [], "licens": [0, 1, 21, 23, 28], "lie": [0, 6, 11, 25, 28, 29, 32], "life": [0, 1, 8, 12, 28], "lifetim": 13, "light": [], "like": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 15, 16, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "likelihood": [0, 1, 5, 9, 28, 29], "lim_": 25, "limit": [0, 5, 6, 8, 12, 22, 23, 28, 29], "lin_clf": 8, "lin_model": [], "lin_reg": 9, "linalg": [0, 2, 5, 6, 8, 11, 13, 17, 22, 25, 28, 29, 30, 31], "line": [0, 3, 6, 8, 11, 13, 15, 16, 20, 28, 30, 31, 32], "line1": 8, "line2": 8, "line2d": [], "line3": 8, "line_model": 15, "line_ms": 15, "line_predict": 15, "linear": [1, 3, 5, 6, 7, 9, 10, 11, 12, 16, 17, 18, 19, 21, 23, 25, 31, 32], "linear_model": [0, 5, 6, 7, 8, 9, 10, 11, 13, 15, 16, 19, 28, 29, 30, 31, 32], "linear_regress": [6, 32], "linearli": [5, 29, 30, 31], "linearloc": [6, 13, 30, 31], "linearregress": [0, 6, 7, 9, 15, 16, 19, 28, 29, 31, 32], "linearsvc": 8, "lineat": 30, "liner": [1, 3], "linerar": 10, "linewidth": [0, 2, 4, 6, 8, 9, 10, 32], "link": [0, 4, 9, 12, 15, 20, 21, 23, 24, 26, 28], "linlag": 5, "linpack": [22, 28], "linreg": [0, 28], "linspac": [0, 2, 3, 4, 6, 8, 9, 10, 13, 16, 17, 19, 22, 25, 28, 29, 31, 32], "linu": 4, "linux": [0, 1, 21, 23, 28], "liquid": [0, 28], "list": [1, 2, 3, 4, 9, 15, 21, 23, 28, 31], "listedcolormap": [9, 10], "literatur": [1, 7, 14, 27, 32], "littl": [1, 3, 9, 12, 31], "live": [8, 16], "ll": [0, 18, 25, 28, 29], "lle": [0, 29], "llm": 20, "lloyd": [4, 14], "lmb": [0, 2, 5, 6, 29, 30, 31, 32], "lmbd": [0, 1, 3, 28], "lmbd_val": [0, 1, 3, 28], "lmbda": [13, 30, 31], "ln": [1, 13, 30], "load": [1, 4, 6, 7, 9, 10, 31], "load_boston": [], "load_breast_canc": [1, 7, 9, 10, 11], "load_data": [3, 4], "load_digit": [1, 3], "load_iri": [8, 9], "loc": [3, 6, 7, 8, 9, 10, 28, 32], "local": [0, 1, 3, 7, 12, 13, 15, 29, 30, 31], "locat": [2, 3, 8, 15], "log": [0, 1, 2, 4, 5, 6, 7, 9, 10, 11, 13, 15, 20, 22, 23, 28, 31, 32], "log10": [0, 5, 6, 29, 30, 31, 32], "log_": [0, 28], "log_clf": 10, "logarithm": [0, 5, 7, 17, 22, 28, 32], "logbook": 23, "logic": [0, 1, 9, 28], "logical_or": [], "login": 15, "logist": [0, 1, 2, 8, 9, 10, 11, 12, 13, 21, 29, 30, 31], "logisticregress": [7, 9, 10, 11], "logit": 7, "logreg": [7, 9, 10, 11], "logspac": [0, 1, 3, 5, 6, 28, 29, 30, 31, 32], "long": [0, 1, 3, 4, 12, 13, 28, 30, 31], "longer": [2, 3, 8, 10, 14, 22, 25, 28, 31], "loocv": [6, 32], "look": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 16, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "loop": [1, 4, 6, 10, 12, 14, 16, 17, 18, 21, 22, 28, 31, 32], "lose": 1, "loss": [0, 1, 3, 4, 5, 6, 7, 8, 10, 11, 13, 18, 22, 23, 28, 32], "loss_fil": 4, "lossfil": 4, "lost": 4, "lot": [1, 4, 6, 16, 19, 20, 31, 32], "low": [0, 6, 9, 10, 11, 23, 28, 29, 32], "lower": [0, 1, 3, 6, 9, 10, 16, 22, 29, 31], "lowercas": [22, 28], "lowest": [9, 13, 25, 31], "lr": [1, 3, 4, 10], "lstat": [], "lstm": 4, "lstm_2layer": 4, "lstsq": [0, 28, 29], "lt": [6, 32], "lu": [0, 5, 28, 29, 30], "lubksb": 22, "luckili": 2, "ludcmp": 22, "lux": 22, "lvert": 1, "lw": [0, 28], "m": [0, 1, 2, 3, 5, 6, 8, 9, 10, 11, 12, 13, 15, 22, 25, 26, 27, 28, 29, 30, 31, 32], "m_": [9, 12], "m_0": 31, "m_1": 14, "m_h": [0, 28], "m_k": 14, "m_l": 12, "m_n": [0, 28], "m_p": [0, 28], "m_t": [13, 31], "ma": 11, "machin": [1, 3, 4, 5, 6, 7, 9, 10, 11, 12, 15, 16, 22, 27, 29, 31, 32], "machinelearn": [0, 6, 16, 20, 21, 23, 24, 26, 27, 28, 29, 30], "machineri": [], "mackai": 27, "macro": [], "made": [0, 1, 3, 4, 5, 6, 7, 9, 11, 12, 23, 28, 29, 31], "mae": [0, 28], "magic": 4, "magnitud": [1, 6, 7, 13, 29, 31], "mai": [0, 1, 2, 3, 5, 6, 7, 8, 9, 11, 12, 13, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "mail": [24, 26], "main": [0, 1, 3, 4, 5, 6, 7, 9, 22, 23, 27, 29, 30, 31], "mainli": [0, 5, 6, 7, 9, 28, 29, 32], "maintain": [6, 31, 32], "major": [1, 6, 9, 10, 13, 22, 28, 30, 31, 32], "make": [1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 15, 16, 18, 19, 21, 22, 23, 25, 27, 28, 30, 31, 32], "make_axes_locat": 6, "make_moon": [8, 9, 10], "make_pipelin": [0, 6, 10, 29, 32], "makedir": [0, 6, 7, 9, 28, 32], "malcondit": 22, "malign": [1, 7, 9], "mammographi": 5, "manag": [0, 2, 3, 15, 21, 23, 28, 31], "mandatori": [26, 28], "mani": [0, 1, 3, 4, 5, 6, 7, 8, 9, 11, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "manifold": 11, "manner": 3, "manual": [6, 29, 31], "map": [0, 1, 2, 6, 7, 8, 11, 12, 14, 25, 28], "marc": 29, "marchant": [], "margin": [0, 5, 8], "marit": [0, 28], "mark": 28, "markdownfil": [], "markdownit": [], "markdownitdeflist": [], "markedli": [], "marker": [7, 22, 28], "markov": [21, 28], "markup": [], "marsaglia": 25, "mask_or": [], "masked_arrai": [], "maskedrecord": [], "mass": [0, 1, 5, 13, 29, 30], "massag": [0, 28], "masses2016": [0, 28], "masses2016ol": [0, 28], "masses2016tre": 0, "masseval2016": [0, 28], "master": [24, 26], "mat": [21, 28], "mat1100": [21, 28], "mat1110": [21, 28], "mat1120": [21, 28], "match": [1, 4, 5, 13, 14, 15, 29, 30, 31], "materi": [4, 5, 7, 13, 15, 22, 24, 26], "math": [3, 7, 12, 13, 22, 25, 27, 28, 31], "mathbb": [0, 4, 5, 6, 7, 8, 11, 12, 13, 14, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "mathbf": [0, 5, 6, 7, 8, 13, 19, 22, 23, 28, 29, 30, 31, 32], "mathcal": [1, 5, 6, 7, 13, 23, 32], "matheemat": 3, "mathemat": [0, 6, 11, 12, 13, 21, 22, 25, 27, 28, 31], "mathemati": 28, "mathrm": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 17, 18, 19, 23, 25, 28, 29, 30, 31, 32], "matmul": [1, 2, 5], "matnat": 27, "matplotlib": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "matplotlibrc": [], "matric": [0, 1, 3, 4, 6, 7, 8, 11, 13, 16, 17, 21, 29, 30], "matrix": [0, 2, 3, 4, 6, 7, 8, 10, 13, 17, 18, 19, 23, 25, 32], "matshow": 1, "matter": [2, 3, 13, 29, 30, 31], "matthia": [], "max": [0, 1, 2, 3, 4, 9, 10, 12, 13, 26, 28, 30, 31], "max_depth": [0, 9, 10], "max_diff": 2, "max_diff1": 2, "max_diff2": 2, "max_it": [0, 1, 8, 13, 28], "max_iter": 14, "max_leaf_nod": 10, "max_sampl": 10, "maxdegre": [0, 6, 10, 29, 32], "maxdepth": 10, "maxim": [1, 4, 5, 7, 8, 11, 32], "maximum": [0, 2, 3, 5, 7, 8, 9, 10, 13, 14, 28, 29, 30, 31], "maxpolydegre": [5, 6, 29, 30, 31, 32], "maxpooling2d": 3, "mbox": [5, 6, 29, 30, 32], "mcculloch": 12, "md": 11, "mdoel": 4, "me": [], "mean": [1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 21, 22, 23, 25, 28, 31, 32], "mean_absolute_error": [0, 28], "mean_divisor": 14, "mean_i": 25, "mean_matrix": 14, "mean_squared_error": [0, 4, 6, 7, 10, 15, 19, 28, 29, 32], "mean_squared_log_error": [0, 28], "mean_vector": 14, "mean_x": 25, "meaning": [0, 4, 7, 28], "meansquarederror": [0, 28], "meant": [3, 7, 10, 13], "meanwhil": 31, "measur": [0, 1, 2, 5, 6, 9, 11, 12, 14, 16, 18, 23, 25, 28, 29, 31, 32], "mechan": [0, 4, 25, 28, 31], "median": [0, 28, 29, 31], "medicin": 12, "medium": [4, 8, 13, 31], "medv": [], "meet": [0, 26], "mehta": [0, 28, 29, 30], "member": 20, "memori": [3, 4, 11, 12, 13, 18, 22], "mentat": [], "mention": [0, 12, 13, 23, 25, 28, 30, 31], "merchant": [], "mere": [0, 23], "merg": [], "meshgrid": [2, 5, 6, 8, 9, 10, 11], "mess": 15, "messag": [5, 13], "messi": 2, "met": [0, 3, 8, 29], "meta": [], "meteorolog": 9, "meter": [6, 29], "method": [0, 1, 2, 3, 4, 5, 7, 8, 11, 12, 14, 15, 16, 17, 18, 19, 20, 21, 22, 25, 27, 29], "metion": 6, "metric": [0, 1, 3, 6, 7, 9, 10, 14, 15, 28, 29, 32], "metropoli": [21, 28], "mev": [0, 25, 28], "mgd": [13, 31], "mglearn": [21, 28], "mgrid": 13, "mhjensen": [], "mi": 10, "mia": [26, 28], "michael": [], "microsoft": 27, "mid": 1, "midel": 4, "midnight": 15, "midpoint": 9, "might": [0, 1, 2, 4, 6, 9, 13, 15, 17, 18, 29, 30, 31], "migth": 17, "mild": 9, "millimet": [6, 29], "million": [0, 28, 29, 31], "mimic": 12, "min": [0, 2, 5, 8, 9, 30], "min_": [0, 2, 5, 14, 17, 28, 29, 30], "min_samples_leaf": 9, "mind": [0, 6, 13, 15, 18, 28, 29, 30, 31, 32], "mindboard": 4, "mine": [21, 28], "mini": [1, 11, 12, 13, 30], "minibatch": [1, 11, 13], "minibathc": [13, 31], "miniforge3": [], "minim": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 29, 30, 31, 32], "minima": [0, 1, 7, 13, 28, 30, 31], "minimum": [0, 1, 2, 6, 8, 9, 11, 13, 29, 30, 31, 32], "minmaxscal": [0, 29, 31], "minor": 25, "minst": 1, "minu": 7, "mirjalili": 28, "mirror": 9, "misc": 6, "misclassif": [8, 9, 10], "misclassifi": [8, 10], "miser": 0, "mismatch": 1, "miss": [7, 10], "mistak": [4, 19], "mit": 27, "mitig": 31, "mix": [1, 2, 28], "mixtur": [13, 31], "mk": [9, 22], "mkdir": [0, 6, 7, 9, 28, 32], "ml": [0, 1, 10, 13, 22, 23, 29, 30, 31], "mlab": 25, "mle": [5, 7], "mlp": 1, "mlpclassifi": 1, "mlpregressor": [0, 28], "mm": 22, "mml": 29, "mn": [12, 25], "mnist": [1, 11], "mo": [], "mod": 25, "mode": [24, 26, 28], "model": [2, 3, 5, 7, 8, 9, 10, 11, 13, 14, 16, 18, 19, 20, 21, 23, 25, 27, 29, 30, 31, 32], "model_select": [0, 1, 3, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "moder": [10, 31], "modern": [0, 6, 7, 21, 28, 31, 32], "modest": 31, "modif": [2, 12, 13], "modifi": [0, 1, 3, 5, 7, 8, 10, 12, 13, 28, 29, 30, 31], "modul": [0, 16, 22, 28], "modular": 25, "modulo": 25, "moe": [11, 29], "moment": [5, 6, 13, 25, 32], "mondai": [26, 28], "monitor": [13, 31], "monoton": [5, 12, 25, 32], "mont": [0, 6, 21, 25, 27, 28, 32], "montli": 16, "moor": [5, 6], "more": [0, 1, 2, 4, 5, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 21, 25], "moreov": [0, 3], "morten": [26, 28, 29, 30, 31, 32], "mortenhj": 28, "most": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 21, 23, 25, 28, 29, 30, 31, 32], "mostli": [1, 11, 18, 31], "motion": [0, 13], "motiv": [1, 4], "moulin": 31, "move": [0, 4, 5, 6, 7, 9, 12, 13, 14, 15, 16, 23, 25, 29, 30, 32], "mpl": [7, 28], "mpl_toolkit": [2, 6, 13, 30, 31], "mplot3d": [2, 6, 13, 30, 31], "mplregressor": 1, "mr_": [], "mrecord": [], "mse": [0, 4, 5, 6, 9, 10, 15, 16, 17, 19, 20, 23, 28, 29, 30, 31, 32], "mse_simpletre": 10, "mselassopredict": [5, 30], "mselassotrain": [5, 30], "mseownridgepredict": [6, 29, 30, 31], "msepredict": [5, 30], "mseridgepredict": [0, 5, 6, 29, 30, 31], "msetrain": [5, 30], "msg": [], "msle": [0, 28], "mt": [7, 12], "mu": [0, 6, 11, 13, 25, 28, 31, 32], "mu0": 25, "mu1": 25, "mu2": 25, "mu_": [6, 25, 29, 31, 32], "mu_i": [6, 29, 31], "mu_n": 11, "mu_x": 25, "much": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 15, 20, 22, 23, 25, 28, 29, 30, 31, 32], "multi": [0, 1, 3, 7, 21, 28], "multiclass": [1, 7], "multidimension": [11, 12, 28], "multilay": 1, "multinomi": 7, "multipl": [2, 4, 5, 6, 7, 12, 13, 15, 25, 29, 30, 31, 32], "multipli": [3, 5, 6, 11, 13, 18, 22, 25, 29, 30, 31], "multiplum": 8, "multivari": [0, 2, 10, 11, 21, 25, 28], "multivariate_norm": [11, 14], "multpli": 16, "murphi": [11, 27, 28], "muse": [], "must": [1, 2, 5, 6, 8, 10, 12, 13, 14, 15, 20, 25, 29, 30, 31, 32], "mutat": 7, "mutual": [1, 3, 6, 13, 32], "mx_": 25, "my": 28, "myenv": [], "myriad": [0, 21, 28], "myself": [], "mz1": 25, "mz2": 25, "m\u00f8svatn": 6, "n": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "n1": 22, "n2": 22, "n8grai": [], "n_": [1, 2, 3, 8, 12, 25], "n_0": [12, 25], "n_boostrap": [6, 10, 32], "n_bootstrap": [6, 32], "n_categori": [1, 3], "n_cluster": 14, "n_compon": 11, "n_epoch": [13, 31], "n_estim": 10, "n_examples_to_gener": 4, "n_featur": [1, 18], "n_filter": 3, "n_hidden": 2, "n_hidden_neuron": [0, 1, 28], "n_i": 25, "n_input": [0, 1, 3, 29], "n_instanc": 9, "n_iter": 31, "n_job": 10, "n_k": 14, "n_l": [12, 25], "n_layer": 1, "n_m": 9, "n_neuron": 1, "n_neurons_connect": 3, "n_neurons_layer1": 1, "n_neurons_layer2": 1, "n_point": 14, "n_sampl": [6, 8, 9, 10, 14, 18, 32], "n_split": [6, 32], "n_step": 4, "n_t": 2, "n_x": 2, "nabla": [1, 13, 30, 31], "nabla_": [2, 13, 30, 31], "nabla_w": 13, "nag": 13, "naimi": [0, 28], "naiv": 7, "naive_kmean": 14, "name": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 15, 18, 20, 21, 22, 23, 25, 26, 28, 29, 30, 32], "namespac": [], "nan": [], "narrow": [13, 31], "nathaniel": [], "nation": [1, 5], "nativ": [21, 28], "natur": [0, 1, 4, 8, 9, 12, 13, 23, 25, 27, 28, 30, 31], "navier": 12, "navig": [15, 31], "nb": 25, "nb_": 22, "nbconvert": 28, "nd": 14, "ndarrai": 6, "ne": [9, 10, 22, 25, 29, 30], "nearest": [1, 3, 6, 11], "nearli": [13, 30], "neat": 28, "neccesari": [6, 32], "necess": 2, "necessari": [0, 1, 3, 4, 8, 14, 18, 28], "necessarili": [0, 4, 11, 25, 28], "necesserali": 5, "neck": 7, "need": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 22, 25, 29, 30, 31, 32], "neg": [0, 1, 3, 5, 6, 7, 10, 13, 22, 25, 28, 30, 32], "neg_mean_squared_error": [6, 32], "neglect": [25, 31], "neglig": 25, "neighbor": [3, 6, 11], "neither": [4, 13, 31], "neq": [13, 14, 25, 30], "nervou": 12, "nest": [9, 12], "nesterov": 13, "net": [2, 4, 12], "netlib": [22, 28], "network": [0, 9, 13, 21, 27, 29], "neural": [0, 13, 21, 27, 29], "neural_network": [0, 1, 2, 28], "neuralnetwork": 1, "neuron": [1, 2, 3, 4, 12], "neutral": [0, 28], "neutron": [0, 28], "never": [1, 4, 6, 9, 25, 32], "new": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 17, 20, 22, 28, 29, 30, 31], "new_chang": [13, 31], "new_hobbit": 28, "new_ma": [], "newaxi": [0, 3, 6, 9, 32], "newli": [0, 28], "newlin": [], "newton": [1, 7, 8, 13, 25], "next": [0, 1, 2, 3, 4, 5, 6, 8, 9, 13, 14, 15, 16, 28, 29, 30, 31, 32], "next_guess": 13, "next_input": 4, "ng": 1, "ni": 14, "nice": [0, 1, 5, 11, 28, 29, 30], "nicer": [18, 31], "nip": 31, "niter": [13, 30, 31], "nitric": [], "nlambda": [0, 5, 6, 29, 30, 31, 32], "nlp": 27, "nm": 25, "nm_n": [0, 28], "nmse": [6, 32], "nn": [2, 5, 6, 12, 22, 28, 32], "nn_model": 1, "nnmin": 2, "node": [1, 3, 9, 10, 12], "nois": [0, 4, 5, 6, 8, 9, 10, 13, 18, 19, 23, 28, 29, 30, 31, 32], "noise_dimens": 4, "noisi": [1, 6, 23, 31, 32], "nomask": [], "non": [0, 1, 3, 5, 6, 7, 9, 10, 11, 12, 13, 14, 18, 22, 25, 28, 29, 30, 32], "nondifferenti": 31, "none": [0, 1, 2, 4, 5, 9, 10, 13, 25, 28, 29], "noninfring": [], "nonlinear": [3, 6, 8, 9, 11, 12, 32], "nonneg": [6, 9, 13, 30, 32], "nonparametr": 6, "nonsens": 25, "nonsingular": 22, "nonumb": [3, 7, 8, 13, 22], "nor": [1, 4, 13, 31], "norm": [0, 1, 5, 6, 8, 11, 13, 18, 28, 29, 30, 31, 32], "normal": [3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 17, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31], "normali": [22, 28], "norwai": [6, 23, 28, 30, 31, 32], "notabl": [], "notat": [0, 2, 5, 6, 13, 14, 25, 28, 29, 30, 32], "note": [0, 1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 14, 15, 16, 18, 21, 22, 25, 27, 28, 31, 32], "notebook": [0, 1, 3, 9, 15, 16, 19, 20, 21, 23, 28, 32], "noteworthi": 31, "noth": [1, 2, 5, 8, 12, 14, 25, 29, 30], "notic": [4, 5, 12, 13, 22, 25, 28], "notion": 3, "novel": [3, 6, 10, 28], "novemb": [1, 26, 28], "now": [0, 2, 4, 5, 6, 7, 8, 10, 11, 12, 14, 15, 16, 19, 21, 22, 23, 25, 28, 29], "nowadai": [0, 1, 3, 9, 21, 28], "nox": [], "np": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 25, 28, 29, 30, 31, 32], "npm": [], "npr": 2, "nsampl": [6, 32], "nt": 2, "nu": 25, "nuclear": [5, 29, 30], "nuclei": [0, 25, 28], "nucleon": [0, 28], "nucleu": [0, 28], "num": 4, "num_coordin": 2, "num_hidden_neuron": 2, "num_it": [2, 18], "num_neuron": 2, "num_neurons_hidden": 2, "num_point": 2, "num_tre": 10, "num_valu": 2, "number": [1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 18, 19, 22, 23, 24, 26, 28, 30, 32], "numberid": 7, "numberparamet": 3, "numer": [0, 5, 6, 9, 10, 11, 12, 13, 21, 22, 27, 28, 29, 30, 31, 32], "numpi": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 21, 23, 25, 29, 30, 31, 32], "numpydocstr": [], "nunmpi": [5, 29], "nve_frngahw": 30, "nx": 2, "ny": 25, "o": [0, 1, 4, 5, 6, 7, 8, 9, 11, 22, 26, 27, 28, 29, 30, 31, 32], "obei": [6, 11, 13, 29, 31], "object": [0, 1, 4, 8, 10, 15, 19, 22, 28, 31], "obliqu": [5, 29, 30], "observ": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 25, 28, 30, 31, 32], "obtain": [0, 1, 5, 6, 7, 8, 9, 10, 12, 13, 14, 17, 22, 23, 25, 28, 29, 30, 31, 32], "obviou": [5, 6, 11, 25, 29, 30], "obviouli": 28, "obvious": [0, 4, 5, 6, 22, 28, 32], "oc": [29, 30], "occupi": [], "occur": [0, 6, 8, 9, 22, 25, 28], "octob": [26, 28], "od": 0, "odd": [0, 3, 7, 28, 29, 31], "odenum": 2, "odesi": 2, "oen": 0, "off": [1, 3, 4, 5, 9, 13, 20, 25, 31, 32], "offer": [6, 11, 21, 22, 24, 26, 28, 32], "offic": [26, 28], "offici": [24, 28], "often": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "ofter": [22, 28], "ol": [0, 13, 17, 19, 29, 31], "old": [1, 5, 10, 13, 15, 18], "old_ma": [], "oliph": [], "ols_paramet": 16, "ols_sk": 6, "ols_svd": 6, "olsbeta": 30, "olstheta": [0, 5], "omega": [2, 3, 6], "omega_0": 3, "omit": [0, 5, 28, 29, 30, 32], "onc": [1, 6, 9, 11, 13, 20, 32], "one": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 19, 20, 21, 22, 23, 25, 26, 28, 29, 31, 32], "onehot": 1, "onehot_vector": 1, "onehotencod": 9, "ones": [0, 2, 5, 6, 8, 9, 10, 11, 13, 16, 18, 22, 23, 28, 29, 30, 31, 32], "ones_lik": 4, "ong": 29, "onl": 3, "onli": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "onlin": [11, 15, 20, 24, 31], "onto": [5, 11, 29, 30], "open": [0, 1, 4, 6, 7, 9, 15, 21, 23, 24, 26, 28, 32], "oper": [0, 1, 3, 5, 6, 10, 11, 12, 13, 15, 16, 21, 25, 28, 29, 30, 31, 32], "operation": 25, "oplu": 25, "opmiz": [13, 31], "opportun": 0, "oppos": [6, 13], "opposit": [1, 5, 8, 29, 30], "opt": [1, 5, 23, 28, 30], "optim": [0, 2, 3, 4, 5, 6, 7, 9, 10, 11, 14, 16, 17, 19, 23, 32], "optimis": [1, 3], "option": [0, 1, 3, 5, 6, 8, 11, 15, 18, 22, 29, 31, 32], "optmiz": [1, 8, 13, 29], "oral": 28, "orang": 0, "order": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 15, 22, 23, 25, 28, 29, 30, 32], "ordinari": [0, 2, 3, 7, 11, 13, 17, 18, 21, 32], "oreilli": [27, 28], "org": [0, 3, 4, 16, 20, 21, 22, 23, 27, 28, 29, 30, 31], "organ": [6, 7, 10, 22, 32], "orient": [1, 5, 25, 29, 30], "origin": [0, 3, 5, 6, 8, 11, 12, 13, 15, 22, 28, 29, 30, 31, 32], "orthogn": [5, 29, 30], "orthogon": [0, 5, 6, 8, 11, 13, 22, 28, 29, 30], "orthonorm": [5, 29, 30], "os": [26, 28], "oscar": 1, "oscil": [3, 13, 31], "oskar": 28, "oskarlei": 28, "osl": 18, "oslo": [0, 21, 23, 24, 26, 28, 29, 30, 31, 32], "osx": [0, 21, 23, 28], "other": [0, 1, 2, 3, 5, 6, 7, 8, 10, 13, 14, 16, 19, 21, 24, 25, 26, 27, 29, 30, 31, 32], "otherwis": [0, 1, 4, 7, 13, 22, 28, 31], "ouput": [5, 7, 12, 32], "our": [1, 2, 3, 6, 7, 8, 9, 10, 12, 14, 15, 16, 17, 18, 19, 21, 22, 25, 31, 32], "ourmodel": 0, "ourselv": [0, 5, 6, 8, 11, 13, 28, 29, 30, 32], "out": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 21, 22, 23, 25, 28, 29, 31, 32], "out_fil": 9, "outcom": [0, 7, 9, 10, 12, 25, 29], "outdoor": 9, "outer": [6, 12, 13], "outfil": 4, "outlier": [0, 8, 28, 29, 31], "outlin": [6, 10, 11, 32], "outlook": 9, "outperform": [10, 31], "output": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 22, 23, 25, 28, 29, 30, 31, 32], "output_bia": 1, "output_bias_gradi": 1, "output_shap": 4, "output_weight": 1, "output_weights_gradi": 1, "outputlayer1": 12, "outputlayer2": 12, "outsid": 4, "over": [0, 1, 3, 4, 5, 6, 9, 10, 12, 13, 15, 16, 19, 22, 23, 28, 29, 30, 31, 32], "over1": 13, "overal": [1, 10, 31], "overcast": 9, "overcom": [12, 13], "overdetermin": [0, 28], "overfit": [0, 1, 3, 6, 9, 10, 13, 31, 32], "overflow": [5, 31, 32], "overhead": 12, "overlap": [3, 7, 8, 9], "overleaf": 20, "overlin": [0, 5, 6, 9, 10, 11, 14, 22, 28, 29, 31], "overshoot": 31, "overst": 0, "overtrain": 4, "overview": [3, 20], "own": [4, 5, 6, 8, 12, 13, 16, 18, 21, 22, 30, 31, 32], "owner": [], "ownmsepredict": 0, "ownmsetrain": 0, "ownridgebeta": 29, "ownridgetheta": [0, 6, 29, 30, 31], "ownypredictridg": 0, "ownytilderidg": 0, "ox": [], "oxid": [], "p": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 25, 28, 29, 30, 31, 32], "p0": 2, "p1": 2, "p_": [2, 4, 8, 9], "p_hidden": 2, "p_i": [5, 25], "p_j": 25, "p_n": 25, "p_output": 2, "p_x": 25, "pack": [0, 28], "packag": [0, 1, 3, 4, 5, 8, 11, 13, 15, 20, 21, 23, 25, 29, 30, 31], "packtpub": 28, "packtpublish": 28, "pad": [3, 4], "page": [0, 21, 28, 30, 31, 32], "pai": [0, 1, 9, 13, 15, 31], "pair": [0, 2, 3, 9, 21, 25, 28], "paltform": 15, "panda": [0, 4, 5, 6, 7, 9, 11, 21, 23, 30, 31, 32], "pandoc": [], "panel": 28, "paper": [1, 31], "paper_fil": 31, "paradigm": [0, 28], "paragraph": 20, "parallel": [10, 13, 21, 22, 28], "param": 2, "paramat": 2, "paramet": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 16, 17, 18, 19, 23, 25, 30, 31, 32], "parameter": [0, 6, 10, 28, 29], "parametr": [0, 6, 28, 29, 32], "paramt": [3, 5, 32], "parser": [], "part": [0, 1, 3, 5, 6, 10, 17, 19, 20, 22, 24, 25, 26, 28, 29, 32], "partial": [0, 1, 5, 6, 7, 8, 10, 11, 12, 13, 16, 25, 28, 29, 30, 31], "particip": [15, 21, 24, 26, 28], "particl": [0, 4, 13, 25, 28], "particular": [0, 1, 2, 3, 5, 6, 9, 10, 11, 12, 13, 16, 23, 25, 27, 28, 29, 30, 31, 32], "particularli": [5, 6, 8, 11, 13, 25, 29, 30, 31, 32], "partit": [1, 4, 9], "partli": [6, 28], "partner": 15, "pass": [2, 3, 12, 14, 31], "password": 23, "past": [10, 25, 31], "patch": [6, 25, 32], "path": [0, 4, 6, 7, 9, 21, 28, 31, 32], "pathcollect": 17, "patholog": [], "patient": 7, "patter": 4, "pattern": [0, 3, 4, 12, 27, 28, 31], "paul": [], "pauli": [0, 28], "pav": [], "pc": [11, 15, 21], "pca": [0, 7, 21, 28, 29], "pd": [0, 4, 5, 6, 7, 9, 11, 28, 29, 30, 31, 32], "pde": 2, "pdf": [0, 3, 4, 5, 6, 9, 15, 16, 19, 20, 23, 27, 28, 32], "pedagog": [0, 28, 29], "penal": [6, 18, 29, 31], "penalti": [6, 13, 18, 23, 29, 31], "penros": [5, 6], "pentagon": [13, 30], "peopl": [1, 9, 13, 21, 31], "per": [0, 1, 6, 24, 26, 28, 31, 32], "percentag": [10, 11, 26], "perceptron": [0, 1, 7, 28], "peregrin": 28, "perez": [], "perfect": [0, 1, 13, 28, 31], "perfectli": [4, 6, 32], "perform": [0, 2, 3, 4, 5, 6, 8, 10, 11, 12, 13, 14, 16, 18, 19, 21, 22, 23, 25, 28, 29, 30, 31, 32], "performac": 4, "perhap": [0, 5, 13, 28, 29, 30, 31], "perimet": 1, "period": [1, 4, 25], "permiss": 15, "permit": [], "permut": 11, "persist": 13, "person": [5, 6, 7, 16, 20, 24, 26, 28, 29], "perspect": 27, "pertin": [12, 28], "petal": [8, 9], "peter": [27, 29], "phantom": 25, "phase": [6, 12], "phenomena": 25, "phenomenon": 31, "phi": 8, "phi_k": 8, "philipp": [], "philosophi": 13, "phone": [26, 28], "photo": [4, 28], "php": 23, "phrase": [0, 28], "physic": [0, 1, 4, 7, 12, 13, 25, 26, 27, 28, 29, 30, 31, 32], "pi": [2, 3, 5, 6, 7, 9, 12, 13, 25, 32], "pick": [1, 9, 10, 11, 13, 14, 31], "pickl": 1, "pictur": [0, 28], "pie": [21, 28], "piec": [11, 14], "pierr": [], "pillow": [0, 21, 23, 28], "pinv": [5, 6, 13, 23, 29, 30, 31], "pip": [0, 1, 15, 21, 23, 28], "pip3": [0, 1, 23, 28], "pipelin": [0, 6, 8, 10, 29, 32], "pippin": 28, "pit": 4, "pitfal": [6, 29], "pitt": 12, "pixel": [1, 3, 4, 28], "pixel_height": [1, 3], "pixel_width": [1, 3], "pkg_resourc": [], "pkgutil": [], "place": [0, 4, 6, 8, 13, 15, 22, 23, 28, 30, 32], "plai": [0, 3, 4, 5, 6, 8, 11, 18, 21, 23, 28, 29, 30, 32], "plain": [8, 10, 12, 13, 14, 30, 31], "plan": [6, 9, 26, 27, 28], "plane": [8, 9], "plateau": [5, 30, 31], "platform": [21, 28], "plausibl": 12, "pleas": [13, 23, 26, 28], "plenti": 1, "plethora": [3, 12], "plot": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31], "plot_all_sc": [23, 29], "plot_confusion_matrix": [7, 10], "plot_count": 6, "plot_cumulative_gain": [7, 10], "plot_data": 1, "plot_dataset": 8, "plot_decision_boundari": [9, 10], "plot_import": 10, "plot_max": 4, "plot_min": 4, "plot_model": 4, "plot_numb": 4, "plot_predict": 8, "plot_regression_predict": 9, "plot_result": 4, "plot_roc": [7, 10], "plot_surfac": [2, 6, 13], "plot_train": 9, "plot_tre": [9, 10], "plt": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 22, 25, 28, 29, 30, 31, 32], "plu": [0, 3, 5, 7, 18, 28, 29], "plugin": [], "pm": [8, 32], "pmatrix": 2, "pml": 27, "pn": 3, "png": [0, 4, 6, 7, 9, 28, 32], "point": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 18, 19, 20, 22, 23, 25, 26, 28, 29, 30, 31, 32], "point_1": 4, "point_2": 4, "poisson": [21, 25, 28], "poli": [6, 8, 32], "poly100_kernel_svm_clf": 8, "poly3": 0, "poly3_plot": 0, "poly_featur": [8, 9, 15], "poly_features10": 9, "poly_fit": 9, "poly_fit10": 9, "poly_kernel_svm_clf": 8, "poly_model": 15, "poly_ms": 15, "poly_predict": 15, "polydegre": [0, 5, 6, 10, 29, 32], "polygon": [13, 30], "polym": 12, "polymi": 23, "polynomi": [0, 5, 6, 7, 8, 9, 10, 11, 15, 17, 19, 20, 23, 28, 29, 31, 32], "polynomial_featur": [6, 15, 16, 17, 32], "polynomial_svm_clf": 8, "polynomialfeatur": [0, 6, 8, 9, 15, 16, 19, 29, 32], "polytrop": [0, 6, 32], "pool": 3, "pool_siz": 3, "poor": [1, 13, 30, 31], "poorli": [0, 29], "popul": [0, 5, 28, 29], "popular": [0, 1, 3, 6, 7, 8, 9, 11, 12, 15, 21, 22, 23, 25, 29], "popularli": [0, 28], "portabl": 10, "portion": [11, 13, 31], "pose": [0, 4, 5, 6, 11, 25, 28, 32], "posit": [0, 1, 2, 3, 5, 7, 8, 10, 11, 13, 14, 22, 25, 28, 29, 30, 31], "possibl": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "possibli": [6, 8, 13, 23], "post": [], "posterior": 5, "postpon": [0, 29], "postscript": 23, "postul": 5, "potenti": [0, 3, 5, 6, 12, 13, 29, 31, 32], "pott": 12, "power": [0, 1, 5, 6, 8, 9, 12, 13, 28, 29, 30, 31, 32], "pp": [5, 6, 19, 32], "practic": [0, 5, 6, 7, 8, 16, 18, 19, 23, 25, 29, 32], "practition": [0, 1, 3, 28, 31], "pre": 28, "preambl": [], "preced": [1, 11, 12, 25], "preceed": 4, "preceq": 8, "precis": [0, 2, 5, 11, 13, 22, 23, 25, 28, 29, 31, 32], "pred": [6, 32], "predicit": 0, "predict": [0, 1, 5, 6, 7, 8, 9, 10, 15, 16, 17, 19, 21, 23, 27, 28, 29, 30, 31, 32], "predict_prob": 1, "predict_proba": [7, 10], "predictor": [0, 5, 6, 7, 9, 10, 11, 28, 29, 31], "prefer": [0, 1, 6, 8, 9, 11, 13, 15, 20, 21, 23, 28], "prefil": [], "prepar": [0, 6, 22, 23, 28, 29], "preprocess": [0, 4, 6, 7, 8, 9, 10, 11, 15, 16, 17, 18, 19, 23, 32], "prerequisit": 0, "prescript": 23, "presenc": 13, "present": [0, 5, 6, 7, 9, 12, 13, 22, 23, 25, 28, 29, 30, 31], "preserv": [3, 11, 22], "press": [13, 15, 27, 30], "pretrain": [1, 4], "pretti": [0, 4, 8, 9, 21, 23, 28], "prettier": [], "prev_centroid": 14, "prevent": [13, 25, 31], "previou": [0, 1, 2, 3, 4, 5, 6, 8, 10, 11, 12, 13, 15, 16, 22, 23, 25, 29, 30, 31], "previous": [2, 3, 9, 10, 25], "price": [0, 4, 9, 13, 31], "primal": 8, "primari": [0, 7, 28], "prime": 25, "princip": [0, 5, 7, 21, 28, 29, 30], "principl": [0, 6, 7, 8, 14, 28, 32], "print": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 18, 22, 25, 28, 29, 30, 31, 32], "print_funct": [8, 9], "printout": [0, 28], "prior": [0, 5, 6, 28], "privat": 0, "prob": [1, 25], "probabilist": [0, 27, 28, 29], "probabl": [0, 1, 3, 4, 6, 7, 10, 13, 21, 28, 29, 31], "problem": [0, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 17, 19, 21, 22, 23, 25, 32], "probml": 27, "proce": [0, 5, 6, 7, 8, 9, 10, 11, 13, 22, 28, 29, 32], "procedur": [2, 4, 5, 6, 8, 10, 11, 13, 29, 30, 31, 32], "proceed": 22, "process": [0, 2, 4, 6, 9, 10, 12, 13, 21, 22, 23, 25, 27, 28, 30, 31, 32], "procur": [], "prod": 27, "prod_": [1, 5, 7, 32], "produc": [0, 3, 4, 5, 6, 9, 10, 11, 12, 13, 18, 20, 21, 22, 25, 28, 29, 32], "product": [0, 1, 3, 5, 6, 7, 8, 12, 13, 16, 17, 21, 22, 28, 29, 31, 32], "profess": [0, 28], "profit": [], "program": [0, 1, 4, 5, 6, 8, 12, 14, 15, 21, 22, 24, 25, 26, 28, 29], "programm": 22, "progress": [1, 4, 14, 31], "prohibit": [6, 32], "project": [0, 1, 2, 3, 5, 11, 13, 15, 19, 21, 24, 29, 30, 31, 32], "project_root_dir": [0, 6, 7, 9, 28, 32], "promin": 12, "promis": 8, "promot": [26, 28], "prompt": 20, "prone": [9, 15], "pronounc": [13, 21, 28, 31], "proof": [0, 11, 12, 13, 28, 30, 32], "prop": 31, "prop_cycl": [], "propag": [2, 3, 13, 31], "proper": [0, 2, 6, 7, 20, 32], "properli": [1, 6, 8, 10, 13, 18, 20, 23, 31], "properti": [0, 1, 3, 12, 13, 16, 22, 28, 32], "proport": [0, 1, 5, 9, 11, 13, 25, 28, 29], "propos": [1, 4, 6, 10, 23, 28, 31], "propto": [5, 13, 30, 31], "proton": [0, 28], "prove": [3, 13, 30, 31], "provid": [0, 1, 3, 4, 5, 6, 8, 9, 10, 12, 13, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "proxi": [1, 13, 31], "prune": 9, "pseudo": [22, 25, 31], "pseudocod": 23, "pseudoinv": 5, "pseudoinvers": [5, 6, 23], "pseudorandom": [6, 25, 32], "psychologi": [0, 28], "pt": 13, "public": [0, 15, 21, 28], "publish": [], "pull": 15, "punish": [0, 1, 28], "pure": [3, 9, 25], "purest": 9, "puriti": 9, "purpos": [0, 3, 10, 12, 14, 28], "push": 15, "put": [1, 20, 31], "putmask": [], "py": 5, "pybtex": [], "pycod": 28, "pydata": 21, "pydevd_extension_api": [], "pydevd_plugin": [], "pydevd_plugin_plugin_nam": [], "pydot": 9, "pygment": [], "pyhton2": 28, "pylab": [7, 28], "pypi": 21, "pyplot": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 19, 22, 25, 28, 29, 30, 31, 32], "pythagora": 5, "python": [1, 2, 3, 5, 6, 8, 11, 12, 13, 14, 18, 20, 23, 25, 29, 31], "python2": [0, 23], "python3": [0, 21, 23, 28], "pythonpath": [], "pytorch": [0, 21, 23, 28], "pyzmq": [], "q": [5, 6, 8, 11, 25, 32], "qp": 8, "qquad": [2, 11, 13, 22, 31], "qr": [5, 6, 22, 29, 30], "quad": [1, 13, 22], "quadrat": [0, 8, 9, 13, 28], "qualit": [4, 9, 23, 25], "qualiti": [0, 9, 21, 28, 29], "quantifi": 1, "quantil": 10, "quantit": [0, 6, 9, 23, 28, 32], "quantiti": [0, 2, 5, 6, 7, 9, 10, 11, 12, 14, 16, 22, 25, 28, 29, 30, 31, 32], "quantum": [4, 12, 27, 28], "quartil": [0, 29, 31], "quench": 5, "queri": 9, "question": [0, 5, 6, 9, 11, 12, 13, 23, 26, 28, 29, 31, 32], "qugan": 4, "quick": [4, 25], "quicker": 31, "quickli": [1, 3, 9, 11, 13, 30, 31], "quit": [1, 5, 6, 9, 10, 12, 15, 29, 30, 32], "quot": 4, "r": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 17, 21, 22, 23, 25, 29, 30, 31, 32], "r2": [0, 5, 6, 19, 28, 29, 30], "r2_score": [0, 28], "r2score": [0, 28], "r_": 31, "r_0": 31, "r_1": 9, "r_2": 9, "r_j": 9, "r_m": 9, "r_t": 31, "rad": [], "rade": [], "radial": [8, 12], "radioact": 25, "radiu": [0, 1, 29, 31], "radziej": [], "ragan": [], "rain": 9, "rais": [], "ram": 31, "ramanujam": [], "ramp": 1, "ran0": 25, "ran1": 25, "ran2": 25, "ran3": 25, "rand": [0, 4, 5, 6, 9, 10, 13, 15, 19, 22, 28, 29, 30, 31, 32], "randint": [6, 9, 13, 31, 32], "randn": [0, 1, 2, 5, 6, 9, 11, 13, 15, 18, 28, 29, 30, 31, 32], "random": [0, 1, 2, 3, 4, 5, 6, 8, 9, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 28, 29, 30, 31, 32], "random_forest_model": 10, "random_index": [13, 31], "random_indic": [1, 3], "random_st": [7, 8, 9, 10, 11], "randomforestclassifi": 10, "randomli": [1, 6, 9, 13, 14, 18, 30, 31, 32], "rang": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 18, 19, 22, 25, 28, 29, 30, 31, 32], "rangl": [0, 6, 11, 25, 28, 29], "rangle_x": 25, "rank": [5, 29, 30], "rankdir": 4, "raphson": [1, 8, 13], "rapidli": [0, 31], "rare": [1, 13, 31], "raschka": [28, 29, 32], "rasckha": 28, "rashcka": [30, 31], "rate": [1, 2, 3, 4, 8, 9, 10, 12, 13, 18, 30, 32], "rather": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 22, 25, 28, 29, 30, 32], "ratio": [4, 7, 9, 10, 11], "rational": [0, 28], "ravel": [5, 6, 7, 8, 9, 10, 11, 13, 22, 32], "raw": [3, 31], "rbf": [8, 11, 12], "rbf_kernel_svm_clf": 8, "rbf_pca": 11, "rc": 25, "rcond": [0, 28, 29], "rcparam": [1, 3, 7, 8, 9, 10, 25, 28], "re": [2, 4, 13, 15, 30], "reach": [1, 4, 5, 6, 9, 10, 12, 13, 14, 30, 31, 32], "react": [], "read": [0, 2, 3, 4, 5, 6, 7, 8, 11, 12, 16, 17, 19, 20, 22, 23, 25, 27, 30], "read_csv": [0, 6, 7, 9, 32], "read_fwf": [0, 28], "reader": [0, 6, 20, 22, 25, 28, 29, 31], "readi": [0, 1, 5, 6, 8, 10, 11, 12, 22, 28], "readili": 1, "readm": [15, 20], "readthedoc": 21, "real": [0, 1, 4, 7, 10, 11, 12, 16, 18, 19, 22, 29, 32], "real_loss": 4, "real_output": 4, "realist": [8, 28], "realiti": 25, "realiz": [1, 12], "realli": [0, 1, 28], "rearrang": 13, "reason": [0, 1, 3, 4, 10, 13, 27, 28, 30, 31], "reassign": 1, "recal": [5, 6, 9, 10, 11, 12, 22, 25, 28, 29, 30, 31, 32], "recarrai": [], "recast": 3, "receiv": [1, 3, 10, 12, 25], "recent": [0, 6, 13, 27, 31, 32], "recept": [3, 12], "receptive_field": 3, "recip": [0, 6, 7, 22, 23, 28, 29], "reciproc": 5, "recogn": [0, 4, 5, 10, 28, 32], "recognit": [0, 1, 3, 12, 27, 28], "recommen": 28, "recommend": [0, 2, 3, 4, 5, 6, 8, 13, 15, 19, 20, 21, 22, 23, 27, 30, 31, 32], "reconsid": 9, "reconstruct": 11, "record": [10, 23, 24, 26, 28], "recreat": 15, "rectangl": [9, 13, 30], "rectangular": [5, 29, 30], "rectifi": [1, 3, 12], "recur": [0, 21, 28], "recurr": [0, 1, 21, 28], "recurs": [9, 21, 22, 28], "red": [0, 3, 4, 6, 8, 9, 31, 32], "redefin": [0, 10, 28, 29, 30], "redefinit": 30, "redistribut": [], "reduc": [1, 3, 5, 6, 9, 10, 11, 13, 28, 30, 31, 32], "reduct": [0, 10, 11, 21, 25, 28, 29], "reegress": 23, "ref": 20, "refer": [0, 1, 2, 3, 5, 6, 11, 12, 13, 14, 20, 22, 27, 28, 29, 30, 31, 32], "referansestil": 20, "referenc": 2, "refin": 12, "refit": [6, 32], "reflect": [0, 1, 4, 5, 23, 25, 28], "refresh": [21, 28], "refreshprogrammingskil": 28, "reg": [10, 11], "regard": [1, 9, 13], "regardless": [12, 16], "regexp": [], "reggi": [], "regim": 31, "region": [3, 4, 6, 9, 12, 23, 31], "regist": [6, 25], "reglasso": [5, 30], "regr_1": [0, 9], "regr_2": [0, 9], "regr_3": [0, 9], "regress": [1, 8, 11, 12, 16, 20, 21, 22], "regressor": [0, 7, 10], "regret": [], "regridg": [0, 5, 6, 29, 30, 31], "regular": [0, 3, 4, 5, 6, 7, 9, 13, 17, 18, 26, 28, 29, 30, 31, 32], "regularli": 15, "reilli": [0, 27, 28], "reinforc": [0, 8, 21, 28], "reiter": 1, "reitz": [], "reject": 7, "rel": [0, 4, 6, 7, 9, 12, 13, 25, 28, 29, 31, 32], "relat": [0, 1, 3, 4, 5, 11, 13, 14, 19, 22, 25, 28, 29, 30, 32], "relationship": [0, 4, 9, 18, 28], "relativeerror": [0, 28, 29], "releas": [1, 21, 28], "relev": [0, 1, 5, 7, 11, 21, 23, 25, 28, 30, 31], "reli": [0, 6, 8, 31], "reliabilti": 23, "reliabl": [7, 25], "relu": [3, 4, 28], "remain": [1, 2, 4, 6, 12, 22, 25, 29, 31, 32], "remaind": 25, "reman": 2, "remark": 1, "rememb": [0, 8, 13, 20, 22, 23, 28, 31], "remind": [0, 5, 11, 13, 19, 22, 25, 32], "remot": 15, "remov": [4, 5, 6, 18, 29, 30, 31], "renam": 15, "render": [0, 28, 29], "reorder": [5, 7, 29, 30], "reorgan": [0, 28], "repeat": [0, 1, 3, 4, 5, 6, 9, 10, 11, 13, 14, 22, 23, 25, 28, 29, 30, 31, 32], "repeated": 28, "repeatedli": [0, 6, 10, 13, 32], "repet": 3, "repetit": [6, 28, 29, 32], "rephras": [13, 30], "replac": [0, 1, 3, 4, 5, 6, 10, 12, 14, 21, 23, 28, 29, 30, 32], "replica": [6, 32], "repo": [15, 23], "report": [28, 31], "repositori": [4, 20, 23, 28], "reposotori": [], "repres": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 23, 25, 28, 29, 30, 31, 32], "represent": [0, 1, 3, 6, 25, 28, 32], "representd": 3, "reproduc": [0, 5, 6, 9, 12, 15, 16, 18, 20, 21, 25, 28, 29], "repuls": [0, 28], "request": [0, 13, 31], "requir": [0, 1, 3, 4, 5, 6, 8, 9, 11, 12, 13, 15, 17, 18, 19, 20, 22, 28, 29, 30, 31, 32], "res1": 2, "res2": 2, "res3": 2, "res_analyt": 2, "res_analytical1": 2, "res_analytical2": 2, "res_analytical3": 2, "resaml": 6, "resampl": [0, 7, 10, 21, 28, 29], "rescal": [0, 11, 12, 31], "rescu": 5, "reseach": 6, "research": [0, 4, 13, 21, 27, 28, 31], "resembl": [6, 25, 32], "reserv": [1, 5, 6, 25, 32], "reshap": [0, 1, 2, 3, 4, 6, 8, 9, 10, 22, 28, 29, 32], "resid": 31, "residenti": [], "residu": [0, 5, 13, 28], "resiz": [5, 29, 30], "resnet": 31, "resort": 31, "resourc": [28, 31], "respect": [0, 1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 13, 14, 16, 17, 18, 23, 25, 28, 29, 30, 31, 32], "respond": 12, "respons": [0, 7, 9, 12, 28, 29], "rest": [0, 5, 18, 29, 30, 31], "restat": [0, 12, 28], "restor": 4, "restored_discrimin": 4, "restored_gener": 4, "restrict": [0, 3, 9, 12, 28], "result": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 28, 31, 32], "retail": [], "retain": [5, 6, 29, 30, 31, 32], "rethink": 32, "return": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 13, 14, 16, 17, 22, 25, 28, 29, 30, 31, 32], "return_data": 14, "return_sequ": 4, "return_x_i": 9, "reus": [1, 3, 6, 19, 20, 23], "reveal": [0, 12, 28], "revers": [1, 22], "review": [21, 22], "revis": [], "revisit": 14, "revolut": 28, "reward": [0, 4, 28], "rewrit": [0, 3, 5, 6, 7, 8, 10, 11, 12, 13, 16, 19, 22, 23, 25, 30, 31], "rewritten": [2, 6, 8, 10, 25, 32], "rewrot": 13, "rf": 10, "rgb": 3, "rgoj5yh7evk": 21, "rh": [6, 32], "rho": [0, 10, 13, 31], "rho_1": 10, "rho_2": 10, "rho_m": 10, "rich": [0, 28], "rid": [], "ride": 9, "rideclass": 9, "ridedata": 9, "ridg": [7, 11, 13, 20, 21, 28, 32], "ridge_paramet": 17, "ridge_sk": 6, "ridgebeta": 30, "ridgetheta": 5, "right": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 12, 13, 14, 16, 17, 19, 22, 23, 25, 28, 29, 30, 31, 32], "right_sid": 2, "rightarrow": [0, 1, 5, 6, 8, 11, 12, 13, 25, 28, 29, 30, 31, 32], "rigor": [0, 28, 29, 30], "ring": 6, "rise": [0, 28], "risk": [0, 13, 28, 30, 31], "rival": 4, "river": [], "rlm": 28, "rm": [25, 31], "rmse": [], "rmsporp": [13, 31], "rmsprop": [1, 3, 4, 13, 23, 32], "rnd_clf": 10, "rng": 25, "rnn": [4, 12], "rnn1": 4, "rnn2": 4, "rnn_2layer": 4, "rnn_input": 4, "rnn_output": 4, "rnn_train": 4, "rntrick1": 25, "rntrick2": 25, "rntrick3": 25, "rntrick4": 25, "ro": [0, 13, 28, 30, 31], "robert": [19, 23, 27], "robust": [0, 28, 31], "robustscal": [0, 29, 31], "roc": [7, 10], "role": [0, 2, 5, 6, 8, 18, 21, 23, 28, 29, 30, 31, 32], "roll": 6, "ronach": [], "room": [0, 26, 28], "root": [0, 5, 9, 13, 15, 25, 29, 30, 31], "root_directori": [], "rot": 28, "rotat": [1, 8, 9, 10], "rotation_matrix": 9, "roughli": [1, 3, 18], "round": [7, 9, 13], "routin": [13, 22, 28, 30], "row": [0, 1, 2, 5, 6, 9, 11, 16, 22, 28, 29, 30, 32], "rr": [5, 29, 30], "rrr": [5, 29, 30], "rubric": [], "rudg": [], "rug": [13, 30, 31], "rule": [0, 1, 5, 6, 13, 23, 28, 29, 30], "run": [0, 1, 2, 4, 5, 6, 8, 9, 11, 13, 15, 20, 21, 23, 28, 29, 30, 31, 32], "runtim": [1, 6, 14, 15], "rust": [0, 21, 22, 28], "rvert": 1, "rvert_2": 1, "s_": [3, 6], "s_1": 6, "s_i": [6, 7], "s_j": 6, "s_k": 6, "s_phenomenon": 23, "saddl": [13, 30, 31], "safeguard": [18, 31], "sai": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 19, 22, 23, 25, 28, 29, 30, 31, 32], "said": [6, 9, 13, 30], "sake": [0, 5, 7, 11, 28, 29, 30], "sale": [0, 28], "sam": 28, "same": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 12, 14, 15, 16, 18, 20, 22, 23, 25, 28, 29, 30], "samm": 10, "sampl": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 14, 18, 19, 21, 22, 23, 25, 28, 29, 31, 32], "sample_vari": 14, "sampleexptvari": 25, "samwis": 28, "sandbox": [], "sasha": [], "sastri": 11, "satisfactori": [0, 28], "satisfi": [1, 2, 3, 6, 8, 13, 22, 25, 30, 32], "satur": [1, 6, 32], "save": [0, 4, 6, 7, 9, 13, 20, 28, 31, 32], "save_fig": [0, 6, 7, 9, 10, 28, 32], "savefig": [0, 4, 6, 7, 9, 25, 28, 32], "savetxt": 4, "saw": [5, 29], "scalabl": 10, "scalar": [2, 5, 6, 10, 29, 32], "scale": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 21, 22, 23, 26, 28, 30], "scale_mean": 4, "scale_std": 4, "scaler": [0, 7, 8, 9, 10, 11, 17, 23, 29], "scan": [5, 7], "scari": 5, "scatter": [0, 1, 6, 7, 8, 9, 14, 15, 17, 28, 29, 31, 32], "scenario": [6, 13, 30, 31], "schedul": [13, 31], "scheme": [1, 13, 30, 31], "schrage": 25, "sch\u00f8yen": [6, 29, 31], "scienc": [0, 1, 10, 12, 13, 21, 24, 25, 26, 27, 30, 32], "scientif": [0, 20, 21, 23, 28], "scientist": [0, 28], "scikit": [3, 5, 6, 8, 9, 10, 13, 15, 16, 20, 21, 22, 23, 27], "scikit_learn": 0, "scikitlearn": 28, "scikitplot": [7, 10], "scipi": [0, 3, 5, 6, 13, 21, 22, 23, 28, 29, 30, 32], "scl": 6, "scm": 15, "score": [0, 1, 3, 6, 7, 9, 10, 11, 15, 16, 19, 23, 26, 28, 29, 31, 32], "scores_kfold": [6, 32], "scratch": [1, 13, 16], "script": [], "sdg": [13, 31], "sdv4f4s2sb8": [30, 31], "seaborn": [0, 1, 3, 6, 7, 28], "seamless": [0, 21, 23, 28], "search": [0, 1, 3, 5, 9, 13, 15, 28, 30, 31], "sebastian": 28, "sebastianraschka": 28, "sec": 6, "second": [0, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 14, 15, 16, 20, 21, 22, 25, 26, 28, 29, 30, 32], "second_mo": 31, "second_term": 31, "secondari": 31, "secondeigvector": 11, "secondli": 12, "section": [4, 11, 16, 20, 22, 23, 25, 29, 31], "sector": 0, "see": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 15, 16, 18, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31, 32], "seed": [0, 1, 2, 3, 4, 5, 6, 8, 9, 11, 13, 14, 18, 20, 25, 28, 29, 30, 31, 32], "seed_imag": 4, "seek": [1, 2, 8], "seem": [1, 3, 4, 31], "seemingli": [0, 28], "seen": [0, 1, 3, 5, 10, 12, 25], "segment": [13, 30], "seismic": 6, "seldomli": [0, 28], "select": [1, 5, 6, 8, 9, 10, 11, 15, 20, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32], "selevet": 15, "self": [1, 5, 29], "sell": 4, "semest": [7, 24], "semi": [8, 13, 30, 31], "semilogx": 6, "send": [5, 12, 13, 26, 28], "senior": [24, 26], "sens": [0, 4, 6, 8, 28, 32], "sensibl": 3, "sensit": [0, 5, 6, 9, 13, 28, 29, 31, 32], "sent": 2, "sentenc": [4, 12], "separ": [0, 1, 2, 4, 6, 8, 9, 12, 14, 18, 21, 23, 25, 28, 31, 32], "septemb": [18, 23, 28], "sequenc": [3, 4, 7, 9, 10, 12, 13, 21, 22, 25, 28, 30], "sequenti": [1, 3, 4, 10, 12, 25], "seri": [0, 1, 2, 3, 4, 5, 6, 10, 11, 12, 13, 22, 28, 29, 30, 32], "serif": [7, 25, 28], "serv": [0, 1, 2, 3, 5, 7, 13, 27, 28, 29, 30, 31], "servic": 23, "session": [1, 15, 20, 23, 24, 26, 28], "set": [1, 4, 5, 6, 7, 8, 10, 11, 13, 14, 16, 17, 18, 21, 22, 23, 25, 26, 31, 32], "set_major_formatt": 6, "set_major_loc": 6, "set_tick": [1, 8], "set_ticklabel": 1, "set_titl": [0, 1, 2, 3, 7, 12, 14, 28], "set_xlabel": [0, 1, 2, 3, 7, 12, 28], "set_xlim": [7, 12], "set_xticklabel": 1, "set_ylabel": [0, 1, 2, 3, 7, 28], "set_ylim": [7, 12], "set_ytick": 7, "set_yticklabel": [1, 6], "set_zlim": 6, "seth": 4, "setminu": 6, "setosa": [8, 9], "setosa_or_versicolor": 8, "setp": [6, 32], "setup": [1, 4, 6, 8, 21, 28, 29, 30], "sever": [0, 3, 5, 6, 7, 8, 9, 11, 12, 13, 16, 21, 22, 23, 25, 28, 29, 30, 31, 32], "sgd": [1, 3, 30], "sgd_clf": 8, "sgdclassifi": 8, "sgdreg": 13, "sgdregressor": 13, "sgn": [5, 29, 30], "shall": [], "shallow": [13, 31], "shape": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 16, 18, 22, 28, 29, 30, 31, 32], "share": [1, 3, 15, 28], "share_mask": [], "shareabl": 15, "she": 7, "sheppard": [], "shibukawa": [], "shift": [1, 6, 12, 15, 18, 25, 29, 31], "ship": 3, "shire": 28, "short": [4, 5, 20, 23], "shortcom": [13, 30, 31], "shorten": 4, "shorter": 25, "shorthand": [28, 32], "shortli": [22, 28], "should": [0, 2, 3, 5, 6, 8, 9, 11, 12, 15, 18, 19, 20, 22, 23, 25, 28, 29, 31, 32], "shouldn": [], "show": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 19, 20, 22, 23, 25, 28, 29, 30, 31, 32], "show_shap": 4, "shown": [0, 4, 5, 8, 12, 13, 22, 29, 30, 31], "shrink": [3, 5, 6, 8, 11, 29, 30, 31], "shrinkag": [5, 6, 29, 30], "shrunk": 11, "shuffl": [0, 1, 4, 6, 13, 29, 31, 32], "side": [0, 2, 5, 8, 12, 13, 22, 23, 28, 30], "sigh": [21, 28], "sigma": [0, 1, 5, 6, 7, 10, 11, 12, 13, 19, 22, 23, 25, 28, 29, 30, 31, 32], "sigma0": 25, "sigma1": 25, "sigma2": 25, "sigma_": [5, 22, 28, 29, 30, 32], "sigma_0": [5, 29, 30], "sigma_1": [5, 29, 30], "sigma_2": [5, 29, 30], "sigma_fn": [7, 12], "sigma_i": [0, 5, 28, 29, 30], "sigma_j": [5, 29, 30], "sigma_m": [6, 25, 32], "sigma_n": [11, 25], "sigma_t": 13, "sigma_x": 25, "sigmoid": [1, 2, 4, 7, 8, 10, 12], "sigmundson": [6, 29, 31], "sign": [1, 2, 7, 8, 10, 25, 26], "signal": [1, 3, 10, 12, 31], "signifi": 4, "signific": [1, 31], "significantli": [1, 13, 18, 25, 30, 31], "sim": [4, 5, 6, 13, 19, 25, 32], "similar": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 14, 18, 21, 22, 23, 28, 30, 32], "similarli": [0, 1, 3, 5, 8, 10, 13, 25, 28, 29, 30, 31], "simpl": [1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 14, 16, 17, 21, 22, 25, 32], "simple_plot": [], "simplepredict": 10, "simpler": [0, 1, 5, 6, 7, 13, 16, 21, 23, 28, 30, 31], "simplernn": 4, "simplest": [0, 1, 3, 4, 9, 10, 12, 14, 23, 28], "simpletre": 10, "simpli": [0, 1, 2, 4, 5, 6, 8, 9, 10, 11, 12, 21, 22, 23, 25, 28, 29, 30, 31, 32], "simplic": [2, 5, 6, 7, 8, 9, 10, 11, 12, 14, 29, 30, 31], "simplicti": [5, 29, 30], "simplifi": [0, 6, 9, 18, 21, 23, 28, 29, 31, 32], "simplist": [3, 6, 25, 32], "simul": [6, 18, 31, 32], "simultan": [6, 31, 32], "sin": [0, 1, 2, 3, 4, 9, 12, 13, 22, 28], "sinc": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 13, 16, 18, 22, 23, 25, 27, 28, 29, 30, 31, 32], "sine": [3, 12], "singl": [0, 1, 2, 3, 5, 6, 7, 8, 9, 12, 13, 18, 19, 22, 25, 28, 29, 30, 31, 32], "singular": [0, 6, 13, 22, 28, 32], "sinusoid": 3, "site": [0, 23, 24, 29], "situat": [0, 4, 5, 7, 13, 25, 28, 29, 30, 31], "six": [3, 25], "size": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 13, 18, 20, 22, 23, 25, 28, 32], "sizesp": 31, "sketch": 10, "ski": 9, "skill": 0, "skip": 11, "skl": [0, 6, 28, 29, 31], "sklearn": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 17, 19, 20, 28, 29, 30, 31, 32], "skplt": [7, 10], "sl": [6, 29, 31], "slack": 8, "slender": [], "slice": [2, 22, 28], "slide": [0, 3, 16, 23, 25, 28, 29, 30], "slight": [6, 13, 32], "slightli": [1, 2, 3, 5, 6, 7, 10, 25, 29, 30, 32], "slope": [8, 11, 12], "slow": [0, 2, 8, 13, 18, 29, 30, 31], "slower": [5, 22, 28, 29, 30, 31], "slowest": 22, "slowli": [12, 31], "slp": 1, "small": [0, 1, 2, 3, 5, 6, 8, 9, 10, 11, 12, 13, 18, 21, 22, 25, 28, 29, 30, 31, 32], "smaller": [0, 1, 2, 5, 6, 8, 9, 11, 13, 25, 28, 29, 30, 31, 32], "smallest": [0, 4, 14, 28], "smallest_row_index": 14, "smodin": [], "smooth": [0, 3, 6, 13, 23, 28, 30, 31], "smoother": 31, "sn": [0, 1, 3, 6, 7, 28], "sne": 11, "so": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 20, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "soar": 6, "social": 0, "soft": [1, 7, 10, 12], "soften": 8, "softmax": [3, 7], "softwar": [0, 8, 21, 22], "sokogskriv": 20, "sol": 8, "sole": [0, 6, 28], "solid": [0, 7], "solut": [0, 1, 2, 3, 5, 6, 8, 10, 11, 13, 18, 22, 23, 25, 28, 29, 30, 31, 32], "solution_ev": 31, "soluton": 2, "solv": [0, 1, 3, 5, 6, 8, 10, 11, 12, 13, 16, 22, 23, 28, 29], "solve_expdec": 2, "solve_ode_deep_neural_network": 2, "solve_ode_neural_network": 2, "solve_pde_deep_neural_network": 2, "solveod": 2, "solveode_popul": 2, "solver": [2, 7, 8, 9, 10, 22, 28], "some": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 15, 16, 18, 19, 23, 25, 28, 31, 32], "some_model": [6, 29, 31], "somehow": 4, "someon": 16, "someth": [0, 1, 3, 4, 7, 9, 11, 15, 19, 20, 23, 25, 28, 29], "sometim": [0, 1, 11, 12, 13, 14, 19, 29, 31], "soon": [22, 26, 29], "sophist": [0, 28], "sopt": 13, "sort": [5, 6, 9, 11, 25, 32], "sound": [3, 5], "sourc": [0, 1, 3, 6, 21, 22, 23, 25, 28, 31, 32], "space": [0, 1, 4, 5, 8, 9, 11, 12, 13, 14, 25, 29, 30, 31], "span": [0, 3, 5, 9, 11, 22, 28, 29, 30], "spare": 1, "spars": [3, 6, 18, 22, 28, 31], "sparse_mtx": [22, 28], "sparsecategoricalcrossentropi": 3, "sparsiti": [10, 18], "spatial": [1, 2, 3, 12], "speak": 25, "special": [6, 7, 10, 12, 13, 22, 25, 28, 29, 30, 31], "specif": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 15, 16, 21, 22, 23, 25, 27, 28, 29, 30, 32], "specifi": [0, 3, 5, 6, 7, 9, 11, 13, 14, 25, 28, 30, 31, 32], "specifici": [0, 10, 28], "spectacular": 3, "spectral": 1, "speech": [0, 1, 3, 4, 12], "speed": [1, 2, 4, 13], "spend": [16, 25, 31], "spent": 23, "sphere": [0, 29, 31], "sphinx": [], "sphinx_book_them": [], "sphinxcontrib": [], "spike": 31, "spin": 6, "spite": 0, "spitzer": [], "spline": 8, "split": [1, 3, 4, 5, 6, 8, 9, 10, 11, 14, 16, 17, 20, 23, 25, 28, 30, 31, 32], "splite": 0, "splitter": [1, 10], "spoiler": [], "spontan": 25, "spot": 3, "spread": [0, 11, 25, 28, 29], "springer": [19, 23, 27, 28, 32], "spuriou": [13, 31], "sqquar": 30, "sqrsignal": 3, "sqrt": [3, 4, 5, 6, 8, 10, 11, 13, 25, 29, 30, 31, 32], "squar": [1, 2, 3, 4, 7, 8, 9, 11, 13, 14, 15, 17, 18, 21, 22, 25, 32], "squarederror": 10, "squaredeuclidean": 14, "squash": 12, "src": [], "srtm": 6, "srtm_data_norway_1": 6, "sso": 20, "stabil": [5, 23, 31], "stabl": [0, 4, 5, 6, 9, 16, 20, 21, 23, 28, 29, 30, 31], "stack": [3, 4], "stage": [5, 13, 15, 23, 31], "stagnat": 31, "stai": [0, 2, 4, 5, 11, 28, 29, 31], "stand": [0, 5, 9, 12, 28, 29, 30], "standard": [0, 1, 4, 5, 6, 7, 8, 10, 12, 17, 18, 19, 22, 23, 25, 28, 30, 31], "standardscal": [0, 6, 7, 8, 9, 10, 11, 17, 29, 31], "standpoint": 31, "stanford": [13, 30], "start": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 22, 25, 26, 28, 29, 30, 31, 32], "start_tim": 14, "starter": [], "stat": [6, 32], "state": [1, 2, 4, 5, 6, 7, 8, 10, 11, 12, 13, 21, 25, 28, 29, 30, 32], "statement": [0, 7, 22, 28], "static": [], "stationari": [30, 31], "statist": [0, 1, 3, 4, 7, 9, 10, 11, 12, 13, 14, 19, 22, 23, 27, 29, 30, 31], "statu": [0, 7, 15, 28], "stavang": 6, "stb": [], "std": [0, 4, 6, 18, 28, 29, 31, 32], "steep": [13, 30, 31], "steepest": 31, "stefan": [], "step": [0, 1, 2, 4, 6, 7, 9, 10, 11, 12, 13, 14, 15, 18, 22, 23, 28, 30], "step_fn": [7, 12], "step_length": [13, 31], "step_siz": 31, "steps_list": 9, "stereo": 3, "sticki": [], "still": [0, 2, 3, 5, 6, 11, 13, 25, 29, 30, 31, 32], "stimuli": 12, "stk": [27, 28], "stk2100": [27, 28], "stk3155": [15, 23, 24, 26], "stk4021": [27, 28], "stk4051": [27, 28], "stk4155": [24, 26], "stk5000": 27, "stochast": [0, 1, 5, 6, 8, 11, 12, 30, 32], "stock": 4, "stoke": 12, "stone": [0, 7], "stop": [1, 4, 9, 13, 14, 18, 30], "storag": [5, 29, 30], "store": [0, 1, 2, 3, 6, 11, 13, 25, 28, 31], "storehaug": [26, 28], "stori": [], "str": [1, 3, 4], "straight": [0, 6, 8, 13, 28, 30, 32], "straightforward": [0, 2, 3, 5, 6, 8, 9, 10, 13, 22, 28, 29, 30, 32], "strategi": [0, 1, 9, 28], "stratifi": [6, 32], "stream": 31, "strength": [0, 5, 14, 29, 30], "stretch": 11, "strict": [8, 13, 30], "strictli": [8, 13, 30], "stride": [4, 22], "strike": 6, "string": 1, "stroke": 7, "strong": [3, 6, 9, 10, 12, 22, 25, 31, 32], "strongli": [0, 8, 15, 20, 21, 22], "stronli": [], "structur": [0, 1, 2, 3, 6, 9, 10, 12, 21, 28, 32], "stuck": [1, 13, 30, 31], "student": [0, 15, 23, 24, 26, 27, 28], "studi": [0, 3, 4, 5, 6, 7, 8, 11, 12, 13, 21, 23, 27, 28, 29, 30, 31], "studier": 27, "style": [7, 9, 20, 22, 28], "stylesheet": [], "st\u00f8land": 26, "sub": [9, 12, 31], "subarrai": [], "subclass": [], "subdivid": [0, 22, 28], "subfield": 0, "subgradi": 31, "subject": [6, 8, 25], "sublicens": [], "sublinear": 31, "submit": 28, "subplot": [0, 1, 3, 4, 6, 7, 8, 9, 10, 14, 28, 32], "subplots_adjust": [8, 25], "subprogram": [22, 28], "subproject": [], "subract": [0, 29], "subroutin": [0, 28], "subscript": 1, "subsequ": [1, 4, 5, 6, 12, 22, 25, 29, 30, 32], "subset": [1, 6, 9, 12, 13, 21, 28, 30, 31, 32], "subspac": [0, 8, 11, 29], "substanti": [9, 10, 31], "substep": 11, "substitut": [3, 6, 12, 16, 22, 32], "subsubset": 9, "subtask": 6, "subtl": 1, "subtract": [0, 4, 5, 6, 11, 13, 18, 19, 22, 23, 25, 29, 31, 32], "subtre": 9, "succeed": [0, 4, 28], "success": [3, 7, 9, 13, 25], "successfulli": [4, 9], "succinctli": 31, "sudo": [0, 21, 23, 28], "suffer": [0, 1, 2, 5, 10, 28, 29, 30], "suffici": [1, 6, 8, 11, 13, 30, 32], "suggest": [1, 13, 23, 27, 30, 31], "suit": [8, 12], "suitabl": [0, 15, 19, 25, 29, 31], "sum": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 19, 22, 25, 28, 29, 30, 31], "sum_": [0, 1, 2, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 19, 22, 23, 25, 28, 29, 30, 31, 32], "sum_i": [0, 2, 5, 6, 8, 13, 19, 23, 29, 30, 31, 32], "sum_j": [6, 18, 31], "sum_ja_": 0, "sum_k": [6, 8, 12, 22], "sum_logist": 13, "sum_m": 3, "sum_n": 3, "sum_nx_": 3, "summar": [5, 6, 9, 32], "summari": [1, 3, 4, 10, 24, 30, 31], "summat": [0, 3, 16, 29, 30], "sunni": 9, "super": [5, 29, 30, 31], "superfici": 3, "superscript": [1, 12], "supervis": [0, 5, 6, 7, 9, 12, 21, 28, 29, 30, 32], "supplement": [7, 23], "suppli": [], "support": [0, 1, 9, 10, 11, 13, 20, 21, 28, 29, 31], "suppos": [0, 5, 6, 7, 8, 10, 11, 12, 13, 22, 28, 29, 30, 31, 32], "suppress": [5, 13, 30], "sure": [0, 1, 4, 6, 16, 20, 23], "surf": 6, "surfac": [0, 6, 28, 31], "surpass": 6, "surpris": [0, 28], "surround": [3, 21], "survei": [0, 5, 6, 28, 29], "svc": [8, 9, 10], "svd": [0, 6, 11, 28, 32], "svdinv": 5, "svm": [8, 9, 10, 11], "svm_clf": [8, 10], "svn": [], "swath": [5, 29, 30], "switch": 0, "sy": [13, 30, 31], "symbol": [1, 5, 11, 13, 21, 25, 28, 29, 30], "symmeteri": 1, "symmetr": [0, 5, 8, 11, 12, 13, 22, 28, 29], "symmetri": 6, "sympi": [0, 21, 23, 28], "synonim": 25, "syntax": 13, "system": [0, 1, 3, 4, 6, 7, 9, 10, 12, 13, 15, 21, 22, 23, 28, 30, 31], "systemat": [4, 6, 32], "t": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 25, 26, 28, 30, 31, 32], "t0": [3, 6, 13, 31], "t1": [2, 13, 31], "t2": 2, "t3": 2, "t9jjwsmsd1o": 32, "t_": 2, "t_0": [2, 9, 13, 31], "t_1": [13, 31], "t_b": 10, "t_i": [1, 2, 5, 12, 29, 30], "t_j": 12, "t_k": 9, "tabl": [9, 23, 25, 26, 28], "tabul": [0, 28], "tabular": 28, "tackl": 4, "tag": [2, 3, 4, 5, 6, 7, 12, 13, 14, 22, 25, 29, 30], "taht": [0, 28], "tail": 25, "tailor": [2, 8, 11, 28], "taiwan": [0, 28], "take": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 17, 19, 21, 22, 25, 28, 29, 30, 31, 32], "taken": [0, 1, 3, 6, 10, 13, 22, 32], "tan": 3, "tangent": [1, 4, 12, 13, 30], "tanh": [1, 4, 7, 8, 12], "target": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 15, 16, 18, 19, 28, 29, 30, 31, 32], "target_nam": 9, "task": [0, 1, 3, 6, 9, 11, 12, 14, 23, 28, 31, 32], "tau": [3, 5, 25], "taught": 28, "tax": [], "taylor": [2, 13, 30], "taylornr": [13, 30], "tc": 8, "teach": [15, 24, 28, 32], "team": 1, "teaser": 0, "technic": [0, 5, 6, 13, 23, 30, 31, 32], "techniqu": [0, 1, 8, 10, 13, 21, 25, 27, 28, 29, 31, 32], "technologi": [0, 1], "tell": [0, 4, 6, 10, 11, 13, 16, 25, 31, 32], "temp": 1, "temp1": 1, "temp2": 1, "temperatur": [0, 9, 28], "templat": [18, 20], "temporari": [], "temporarili": 1, "ten": [3, 28], "tend": [3, 5, 6, 8, 9, 10, 12, 13, 14, 29, 31, 32], "tendenc": [0, 28], "tension": [6, 32], "tensor": 3, "tensorflow": [0, 2, 4, 8, 14, 21, 22, 23, 27, 28, 29], "term": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 19, 23, 25, 28, 29, 30, 31], "term1": [5, 6, 11], "term2": [5, 6, 11], "term3": [5, 6, 11], "term4": [5, 6, 11], "termin": [0, 4, 5, 9, 10, 13, 15, 29, 30, 31], "terminarl": 15, "terrain": 6, "terrain1": 6, "test": [3, 4, 5, 6, 7, 8, 9, 10, 13, 16, 19, 20, 22, 23, 25, 28, 30, 31, 32], "test_acc": 3, "test_accuraci": [1, 3], "test_error": 6, "test_imag": [3, 4], "test_ind": [6, 32], "test_input": 4, "test_label": [3, 4], "test_loss": 3, "test_pr": 1, "test_predict": 1, "test_rnn": 4, "test_scor": [7, 10], "test_siz": [0, 1, 3, 5, 6, 10, 15, 17, 29, 30, 31, 32], "test_split": 9, "testerror": [0, 6, 29, 32], "testi": 4, "testpredict": 4, "testx": 4, "tex": [], "text": [0, 1, 2, 4, 5, 8, 9, 11, 13, 15, 18, 20, 22, 25, 27, 29, 30, 31, 32], "textbf": [], "textbook": [16, 23, 29, 30, 32], "textual": 9, "textur": 1, "tf": [1, 3, 4, 13, 14, 30], "th": [0, 1, 2, 5, 6, 7, 9, 12, 13, 14, 22, 23, 25, 28, 29, 31, 32], "than": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 17, 21, 25, 28, 29, 31, 32], "thank": [4, 6, 29, 31], "theano": [1, 21, 28], "thei": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 16, 18, 20, 22, 23, 25, 28, 29, 30, 31, 32], "them": [0, 1, 3, 4, 6, 8, 9, 10, 11, 12, 13, 18, 22, 23, 28, 29], "theme": [0, 15, 28], "themselv": [0, 23, 25, 28, 31], "thenc": [6, 32], "theorem": [2, 6, 7, 29, 30], "theoret": [0, 4, 10], "theori": [0, 1, 3, 8, 9, 12, 13, 19, 21, 23, 27, 28, 31], "thereaft": [0, 5, 6, 11, 12, 22, 23, 28, 32], "therebi": [0, 5, 7, 11, 23, 28, 29, 30], "therefor": [0, 1, 2, 3, 4, 6, 7, 8, 11, 13, 19, 25, 28, 29, 30, 31, 32], "therein": 11, "thereof": [0, 6, 13, 28, 31, 32], "theta": [0, 1, 4, 5, 6, 7, 13, 16, 23, 25, 28, 29, 30, 31], "theta1": 31, "theta2": 31, "theta_": [0, 1, 6, 7, 13, 28, 29, 30, 31], "theta_0": [0, 5, 6, 7, 16, 28, 29, 30, 31], "theta_0x_": [0, 28, 29], "theta_1": [0, 5, 6, 7, 28, 29, 30, 31], "theta_1x_": [0, 28, 29], "theta_1x_0": [0, 28], "theta_1x_1": [0, 7, 28], "theta_1x_2": [0, 28], "theta_1x_i": [7, 29, 30, 31], "theta_2": [0, 28, 29], "theta_2x_": [0, 28, 29], "theta_2x_0": [0, 28], "theta_2x_1": [0, 28], "theta_2x_2": [0, 7, 28], "theta_2x_i": 29, "theta_3x_i": 29, "theta_4x_i": 29, "theta_closed_form": 18, "theta_closed_formol": 18, "theta_closed_formridg": 18, "theta_gdol": 18, "theta_gdridg": 18, "theta_i": [0, 1, 5, 28, 29, 30], "theta_j": [0, 5, 6, 18, 28, 29, 31], "theta_k": [30, 31], "theta_linreg": [13, 30, 31], "theta_ol": 18, "theta_p": 7, "theta_px_p": 7, "theta_ridg": 18, "theta_t": [13, 31], "theta_tru": 18, "thetaith": 31, "thetavalu": 5, "thi": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 29, 30, 31, 32], "thing": [0, 1, 2, 4, 5, 7, 9, 15, 16, 18, 25, 28, 32], "think": [0, 1, 3, 4, 6, 9, 12, 13, 14, 25, 28, 29, 30, 31, 32], "third": [0, 3, 6, 13, 26, 28, 30, 31], "thirti": 7, "thorughout": 28, "those": [0, 3, 5, 6, 8, 9, 10, 11, 22, 23, 28, 29, 30, 31, 32], "though": [1, 2, 3, 4, 13, 16, 17, 19, 22, 25, 31], "thought": [6, 14, 23, 25, 32], "thousand": [0, 1, 23, 29, 31], "three": [0, 1, 3, 5, 6, 8, 9, 12, 22, 23, 24, 25, 26, 28, 29, 30, 32], "threshold": [1, 3, 9, 10, 11, 12, 13, 31], "through": [0, 1, 2, 3, 4, 5, 6, 8, 11, 12, 13, 14, 15, 21, 22, 23, 25, 28, 29, 30, 31, 32], "throughout": [0, 4, 5, 14, 15, 21, 22, 25, 28], "throw": [3, 6, 25, 32], "thu": [0, 1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 26, 28, 29, 30, 31, 32], "thumb": [0, 6, 23, 29], "thursdai": [], "tibshirani": [6, 19, 23, 27, 28, 32], "tick_param": 6, "ticker": [6, 13, 25, 30, 31], "tif": 6, "tight_layout": [1, 7], "tightli": 11, "tild": [0, 5, 6, 7, 11, 19, 23, 25, 28, 29, 30, 31, 32], "till": [0, 4, 7, 8, 9, 10, 12, 22, 28, 29], "time": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 20, 21, 22, 23, 25, 28, 29, 30, 32], "timeit": 4, "timer": 4, "timeseri": [], "tini": [1, 31], "tip": 3, "titl": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 13, 15, 20, 25, 28, 30, 31, 32], "tm": [], "tmp": 13, "tn": [2, 3, 7], "to_categor": [1, 3, 4], "to_categorical_numpi": 1, "to_numer": [0, 6, 28, 32], "todai": 3, "togeth": [0, 3, 6, 8, 11, 13, 21, 28], "toi": 14, "token": [], "told": 13, "toler": [2, 14], "tolist": 4, "tomographi": 12, "too": [0, 2, 4, 5, 6, 9, 11, 13, 17, 18, 25, 27, 29, 30, 31, 32], "took": [8, 28], "tool": [0, 1, 3, 6, 13, 15, 21, 29, 32], "toolbox": 8, "top": [0, 3, 5, 6, 9, 10, 19, 21, 28, 32], "topic": [0, 5, 6, 7, 8, 21, 23, 29, 30, 32], "topolog": [3, 12], "topologi": [1, 12], "torkjellsdatt": [26, 28], "tort": [], "toss": [10, 25], "total": [0, 1, 2, 3, 4, 6, 7, 8, 10, 11, 12, 13, 14, 22, 25, 26, 28, 29, 30, 31, 32], "total_loss": 4, "totalclustervari": 14, "totalscatt": 14, "toward": [1, 2, 7, 12, 13, 15, 30], "towardsdatasci": 31, "town": [], "tp": [4, 7], "tpng": 9, "tpu": [13, 21, 28], "tqdm": 6, "tr": [], "track": [3, 13, 14, 15, 22, 29, 30, 31], "tract": [], "tractabl": [0, 28, 29], "trade": [5, 9, 20, 31, 32], "tradeoff": [0, 5, 19, 23, 28, 29, 30], "tradit": [0, 1, 4, 6, 28, 32], "train": [2, 3, 5, 6, 8, 9, 10, 11, 12, 13, 16, 17, 20, 23, 30, 31, 32], "train_accuraci": [0, 1, 3, 28], "train_dataset": 4, "train_end": [0, 1, 29], "train_error": 6, "train_imag": [3, 4], "train_ind": [6, 32], "train_label": [3, 4], "train_pr": 1, "train_siz": [0, 1, 3, 29], "train_step": 4, "train_test_split": [0, 1, 3, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "train_test_split_numpi": [0, 1, 29], "trainable_vari": 4, "trained_model": [6, 29, 31], "trainerror": [0, 29], "traini": 4, "training_checkpoint": 4, "training_dataset": 4, "training_gradi": [13, 31], "trainingerror": [6, 32], "trainpredict": 4, "trainscor": 4, "trainx": 4, "trait": [0, 28], "trajectori": [4, 31], "transfer": [9, 28], "transform": [0, 5, 6, 7, 8, 9, 10, 11, 12, 13, 17, 21, 22, 28, 29, 30, 31, 32], "transit": [6, 12], "translat": [1, 4, 6, 10, 28, 29, 31], "transpos": [1, 5, 11, 22, 29, 30], "travers": [0, 5], "travi": [], "treat": [0, 1, 3, 6, 12, 13, 18, 25, 28, 29, 30, 31, 32], "tree": [0, 1, 21, 28], "tree_clf": [9, 10], "tree_clf_": 9, "tree_clf_sr": 9, "tree_reg": 9, "tree_reg1": 9, "tree_reg2": 9, "trend": 25, "treue": 7, "trevor": [19, 23, 27], "tri": [2, 3, 4, 9, 13, 16, 31], "triain": 0, "trial": [0, 2, 4, 6, 13, 25, 28, 30, 31, 32], "triangl": [13, 30], "triangular": 22, "trick": [3, 4, 8, 11, 13, 25, 31], "trickier": 25, "tridiagon": 22, "trillion": 21, "trim": [], "trivial": [0, 1, 5, 11, 25, 28, 30], "troffa": [], "troubl": [0, 8, 12, 15, 29, 31], "truck": 3, "true": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 14, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "true_beta": 29, "true_fun": [6, 32], "true_theta": [6, 31], "truli": 28, "try": [0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 13, 14, 15, 18, 21, 22, 23, 25, 28, 29, 30, 31], "tr\u00f6ger": [], "tucker": 8, "tuesdai": [26, 28], "tumor": [7, 9], "tumour": 7, "tunabl": 1, "tune": [4, 9, 13, 22, 28, 31], "turn": [0, 1, 5, 6, 7, 8, 9, 10, 11, 12, 13, 22, 23, 25, 28, 29, 30, 31, 32], "tutori": [1, 4], "tv": 2, "tveito": 2, "tweak": [1, 4, 10, 25], "twice": [13, 30], "twist": 11, "two": [0, 1, 2, 4, 5, 6, 7, 9, 10, 11, 12, 13, 15, 17, 22, 23, 24, 25, 27, 28, 29, 30, 31, 32], "tx": [13, 30, 31], "tx_1": [13, 30], "txt": [4, 15, 20], "ty": [13, 30], "type": [0, 1, 3, 6, 8, 10, 13, 22, 25, 29, 30, 31, 32], "typeset": 20, "typic": [0, 1, 2, 3, 4, 5, 7, 9, 10, 12, 13, 15, 16, 20, 25, 28, 29, 30, 31, 32], "typo": 23, "u": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 22, 23, 25, 27, 28, 29, 30, 31, 32], "u_": 22, "u_i": 12, "u_m": 10, "ua": [0, 28], "ubuntu": [0, 21, 23, 28], "uci": 23, "ufunc": [], "uio": [15, 20, 23, 26, 27], "uk": [], "un": 14, "unabl": 15, "unari": [22, 28], "unbalanc": [6, 9, 32], "unbias": [0, 5, 6, 28, 32], "uncent": [6, 29, 31], "uncertainti": [0, 5, 28], "uncertitud": 25, "unchang": [1, 3], "uncom": [], "uncorrel": [10, 25], "undefin": [5, 29, 30], "under": [0, 1, 5, 6, 10, 13, 21, 23, 28, 29, 30, 31, 32], "underdetermin": [0, 28], "underfit": [1, 6, 32], "underflowproblem": [5, 32], "undergo": 5, "undergradu": [24, 26], "underli": [0, 1, 9, 13, 18, 25, 28, 31], "underlin": [], "underscor": [], "underset": [4, 14], "understand": [0, 1, 3, 5, 6, 10, 13, 14, 15, 19, 20, 21, 28, 29, 30, 31], "understood": [8, 13], "underwai": [], "undesir": 8, "undetermin": [5, 8, 32], "undo": 4, "unexpect": [6, 32], "unexpected": 25, "unexplain": 18, "unfair": [6, 29], "unfortun": [1, 8, 9, 10], "unicode_liter": [8, 9], "uniform": [0, 1, 5, 6, 11, 13, 23, 25, 28, 30, 31], "uniformli": [13, 25, 30, 31], "unifrompdf": 25, "unimport": [13, 30], "union": [5, 6, 32], "uniqu": [0, 2, 6, 13, 14, 22, 28, 32], "unique_cluster_label": 14, "unit": [0, 1, 3, 4, 5, 10, 12, 18, 25, 28, 29, 30, 31], "unitari": [5, 6, 22, 29, 30], "unitarili": [22, 28], "uniti": 25, "univari": 25, "univers": [0, 1, 2, 13, 21, 23, 24, 26, 28, 29, 30, 31, 32], "unix": 1, "unknow": [0, 22, 28], "unknown": [0, 1, 3, 4, 5, 6, 8, 10, 13, 22, 23, 28, 29, 30, 31, 32], "unknowwn": 12, "unlabel": 1, "unless": [0, 3, 6, 11, 13, 23, 28, 30, 32], "unlik": [1, 3, 8, 13, 30, 31], "unnecessarili": 9, "unord": 3, "unpickl": [], "unpleas": [], "unpublish": 31, "unravel": 1, "unrol": [3, 11], "unscal": 19, "unseen": [0, 7, 9, 15], "unstabl": 1, "unsupervis": [0, 1, 4, 12, 21, 28], "unsymmetr": [22, 28], "until": [1, 2, 4, 9, 12, 13, 14, 30, 31], "untouch": 0, "unusu": 12, "up": [1, 3, 4, 5, 6, 8, 10, 11, 13, 14, 16, 18, 19, 20, 21, 22, 23, 25, 26, 31], "updat": [1, 2, 10, 12, 13, 14, 15, 18, 32], "uploa": 28, "upload": [15, 20, 21, 23, 27], "upon": [0, 1, 6, 7, 11, 22], "upper": [0, 8, 9, 16, 22, 29], "uppercas": [22, 28], "upsampl": 4, "upscal": 4, "upward": [], "url": [28, 29], "us": [4, 5, 6, 8, 9, 10, 11, 12, 14, 15, 17, 20, 22, 25, 27, 32], "usag": [0, 8, 21, 28, 29], "usd": [], "usd10000": [], "use_bia": 4, "usecol": [0, 28], "useless": 1, "user": [0, 1, 2, 4, 6, 7, 15, 21, 22, 23, 28, 29], "usernam": [15, 23], "usetex": 25, "usg": 6, "usr": 25, "usual": [0, 3, 4, 7, 12, 13, 14, 28, 31], "ut": 5, "utf": [], "util": [1, 3, 4, 6, 7, 10, 14, 19, 28, 32], "ux": 22, "v": [2, 4, 5, 6, 11, 13, 15, 21, 29, 30, 32], "v0": 25, "v1": 25, "v2": 25, "v5": [], "v_": 31, "v_0": [11, 31], "v_t": 31, "va": 1, "vahid": 28, "val": 13, "val_accuraci": 3, "val_loss": 4, "vale": 2, "valid": [0, 1, 4, 7, 9, 10, 13, 21, 25, 28, 29, 31], "validation_data": 3, "validation_split": 4, "valu": [0, 1, 2, 3, 4, 6, 7, 8, 9, 10, 12, 13, 14, 16, 17, 18, 20, 21, 22, 23, 28, 31], "valuat": 9, "valueerror": [], "valy": 4, "van": [0, 19, 23, 28, 29, 30, 31], "vandenbergh": [8, 13, 30], "vandermond": [0, 28], "vanilla": [0, 6, 11, 14, 29, 31], "vanish": [1, 4, 13, 25, 30], "var": [5, 6, 10, 11, 19, 23, 25, 29, 32], "var_x": 25, "varabl": 8, "varepsilon": [5, 6, 19, 32], "varepsilon_": [5, 6, 32], "varepsilon_i": [5, 6, 32], "vari": [0, 1, 3, 5, 6, 10, 28, 32], "variabl": [0, 1, 2, 5, 6, 7, 8, 10, 11, 12, 13, 14, 22, 28, 29, 31, 32], "varianc": [0, 1, 5, 7, 9, 10, 11, 13, 14, 18, 20, 21, 22, 25, 28, 29, 30, 31], "variance_i": [5, 11, 29], "variance_x": [5, 11, 29], "variant": [0, 1, 6, 8, 12, 13, 28, 29, 30, 31], "variat": [3, 4, 11, 28], "varieti": [0, 3, 12, 21, 23, 28], "variou": [1, 3, 5, 6, 7, 8, 9, 11, 12, 13, 16, 19, 20, 21, 22, 23, 25, 28, 29, 30, 31], "varydimens": 4, "vast": 31, "vastli": 3, "vaue": 1, "vault": 0, "vdot": [2, 13, 30, 31], "ve": [23, 31], "vec": [6, 32], "vector": [0, 1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 13, 14, 17, 18, 21, 30, 31, 32], "vector_mean": 14, "ventur": [0, 8, 21, 28], "venv": 15, "verbos": [1, 3, 4], "veri": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 23, 25, 27, 28, 29, 30, 31, 32], "verifi": [3, 11, 22, 28], "versatil": [8, 28], "versicolor": [8, 9], "version": [0, 3, 10, 13, 14, 15, 21, 22, 23, 25, 28], "versu": [1, 31], "vert": [0, 1, 5, 6, 7, 8, 9, 11, 13, 16, 17, 28, 29, 30, 31, 32], "vert_1": [5, 6, 29, 30, 31], "vert_2": [5, 6, 11, 17, 29, 30, 31, 32], "via": [0, 5, 6, 7, 8, 9, 10, 11, 12, 19, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32], "vidal": 11, "video": [0, 1, 12, 21, 24, 26, 28, 29, 30], "view": [1, 3, 5, 6, 12, 13, 25, 27, 28, 30, 31, 32], "violat": 8, "virginica": 9, "viridi": [0, 1, 2, 3, 28], "virtanen": [], "virtual": [1, 31], "viscos": 13, "viscou": 13, "visibl": 15, "vision": [0, 3], "visit": 31, "visual": [0, 3, 11, 12, 18, 21, 28, 29], "visualis": 1, "visualstudio": [15, 16, 19], "viz": [6, 8, 25], "vmap": 13, "vmax": [1, 6], "vmh0zpt0tli": 31, "vmin": [1, 6], "voic": 3, "volatil": 31, "volum": [0, 3, 28], "vote": [10, 28], "voting_clf": 10, "votingclassifi": 10, "votingsimpl": 10, "vscode": [], "vstack": [5, 11, 22, 25, 28, 29], "vt": [5, 29, 30], "w": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 14, 22, 25, 28, 29, 30, 31, 32], "w1": 8, "w2": [8, 11], "w3": 8, "w_": [1, 12], "w_1": [8, 22], "w_1x_": 8, "w_1x_1": 8, "w_2": [8, 22], "w_2x_": 8, "w_2x_2": 8, "w_3": 22, "w_4": 22, "w_hidden": 2, "w_i": [1, 2, 10], "w_ix_i": 12, "w_j": 22, "w_m": 22, "w_output": 2, "w_px_": 8, "w_px_p": 8, "w_t": [], "wa": [1, 3, 4, 5, 6, 7, 10, 11, 12, 14, 17, 22, 28, 29, 31, 32], "wai": [0, 1, 2, 3, 4, 5, 6, 7, 8, 10, 11, 12, 13, 14, 15, 18, 19, 22, 25, 28, 29, 30, 31], "walk": 9, "walker": 25, "wall": 31, "walt": [], "wang": [0, 28], "want": [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 20, 21, 23, 25, 28, 29, 30, 31, 32], "warn": 4, "warrant": [6, 32], "warranti": [], "wast": [3, 31], "watch": [21, 30, 31, 32], "wave": 3, "wavelet": 8, "wcag": [], "we": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27, 29, 30, 32], "weak": [9, 10, 14], "weaker": 31, "weather": [1, 12], "web": [21, 24, 26, 28], "webpag": 28, "websit": [6, 22, 23, 24, 28], "wedg": [8, 25], "wednesdai": [26, 28], "wee": 11, "week": [0, 5, 6, 7, 23, 24, 26], "weekli": [15, 16, 21, 23, 24, 26, 27, 28], "weight": [1, 2, 3, 6, 7, 9, 10, 12, 13, 18, 25, 31], "weigth": 2, "welcom": [8, 15, 21], "well": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 12, 13, 15, 16, 20, 21, 22, 23, 25, 27, 28, 29, 30, 31, 32], "went": 8, "were": [0, 1, 3, 4, 5, 6, 7, 8, 10, 11, 12, 14, 25, 28, 31, 32], "wessel": [0, 19, 23, 28, 29, 30, 31], "what": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 19, 20, 21, 22, 23, 25, 31], "whatev": 3, "when": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 18, 19, 22, 23, 25, 28, 29, 30, 32], "whenev": [13, 15, 25, 31], "where": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 20, 21, 22, 23, 25, 26, 28, 29, 30, 31, 32], "wherea": [6, 25, 31, 32], "wherefrom": 23, "wherein": [1, 12], "whether": [0, 3, 5, 7, 9, 23, 25, 28], "which": [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 32], "whichev": [1, 3], "while": [0, 1, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 15, 16, 19, 20, 25, 28, 29, 30, 31, 32], "white": 9, "whiteboad": 31, "whiteboard": [29, 30, 31], "who": [0, 15], "whole": [1, 3, 4, 5, 9, 11, 13, 31], "whom": [], "whose": [0, 6, 10, 25, 29, 32], "whow": [11, 29], "why": [0, 1, 3, 6, 13, 15, 16, 17, 19, 23, 29, 30], "wide": [0, 1, 3, 6, 7, 12, 21, 22, 23, 28, 32], "widehat": [6, 32], "width": [0, 3, 8, 9, 28], "wieringen": [0, 19, 23, 28, 29, 30, 31], "wiki": 23, "wikipedia": 23, "win": [10, 31], "wind": 9, "window": [], "wing": [26, 28], "winther": 2, "wiothout": 6, "wiscons": 7, "wisconsin": 10, "wisdom": [6, 29, 31], "wise": [1, 5, 12, 13, 29, 30, 31], "wish": [0, 2, 5, 7, 8, 11, 13, 14, 18, 22, 23, 28, 29, 30, 31], "with_std": [0, 29], "wither": 6, "within": [0, 2, 3, 4, 7, 9, 12, 13, 14, 25, 27, 28, 30], "withinclust": 14, "without": [0, 1, 5, 6, 8, 9, 11, 12, 13, 15, 18, 23, 28, 29, 30, 31, 32], "won": [0, 15, 28], "wonder": 8, "word": [0, 1, 3, 4, 5, 6, 7, 14, 19, 25, 28, 29, 30, 31], "work": [0, 1, 4, 6, 7, 8, 9, 13, 15, 16, 18, 19, 20, 21, 23, 24, 25, 26, 28, 29, 31, 32], "workabl": 31, "workaround": [], "workhors": 31, "workload": 31, "workshop": 28, "world": [0, 8, 16, 29], "worldwid": [0, 28], "worri": 15, "wors": [0, 1, 3, 4, 6, 28, 31, 32], "worth": [9, 19], "worthi": 23, "would": [0, 1, 3, 5, 6, 7, 8, 9, 10, 11, 12, 13, 16, 18, 20, 22, 23, 25, 28, 29, 30, 31, 32], "wouldn": [], "wrap": [6, 22, 28], "write": [0, 1, 2, 3, 5, 6, 7, 8, 12, 13, 15, 16, 18, 22, 28, 29, 31, 32], "written": [0, 2, 3, 5, 11, 12, 13, 16, 21, 22, 23, 25, 28, 29, 30, 31], "wrong": [1, 8, 15], "wrongli": 10, "wrote": [5, 11, 29], "wrt": [10, 13, 31], "wth": [10, 13, 31], "www": [20, 21, 22, 23, 27, 28, 30, 31, 32], "wx_1": 8, "x": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 28, 30, 31, 32], "x0": 8, "x1": [4, 8, 9, 10, 13], "x1_exampl": 8, "x1d": 8, "x2": [8, 9, 10, 13], "x2d": [8, 11], "x2d_train": 11, "x2dsl": 11, "x3": 8, "x_": [0, 2, 3, 5, 6, 8, 10, 11, 13, 14, 22, 25, 28, 29, 30, 31, 32], "x_0": [0, 5, 11, 18, 22, 28, 29, 32], "x_1": [0, 2, 5, 6, 7, 8, 9, 10, 11, 13, 18, 22, 25, 28, 29, 30, 31, 32], "x_2": [0, 2, 5, 6, 7, 8, 9, 10, 11, 13, 22, 25, 28, 29, 30, 32], "x_3": [8, 22, 25], "x_4": 22, "x_6": 18, "x_center": 11, "x_data": 1, "x_data_ful": 1, "x_hidden": 2, "x_i": [0, 1, 2, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 22, 25, 28, 29, 30, 31, 32], "x_input": 2, "x_ix_": [0, 28], "x_iy_i": 8, "x_j": [0, 2, 8, 9, 12, 16, 25, 29, 31], "x_jy_j": 8, "x_k": [12, 14, 22, 25, 29], "x_l": 25, "x_m": [6, 12, 22, 25, 32], "x_mean": [18, 31], "x_n": [0, 2, 3, 6, 8, 11, 12, 13, 22, 25, 28, 30, 32], "x_new": [9, 10], "x_norm": [18, 31], "x_offset": [6, 29, 31], "x_output": 2, "x_p": [3, 7, 9], "x_poli": 9, "x_poly10": 9, "x_pred": 4, "x_prev": 2, "x_reduc": 11, "x_sampl": [], "x_scale": 8, "x_small": 13, "x_std": [18, 31], "x_t": 31, "x_test": [0, 1, 3, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 29, 30, 31, 32], "x_test_": 17, "x_test_own": 6, "x_test_scal": [0, 6, 7, 9, 10, 11, 29, 31], "x_tot": 4, "x_train": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "x_train_": 17, "x_train_mean": [6, 29, 31], "x_train_own": 6, "x_train_r": 19, "x_train_scal": [0, 6, 7, 9, 10, 11, 29, 31], "x_val": 1, "xarrai": [21, 28], "xavier": 1, "xbnew": [13, 30, 31], "xcode": [0, 21, 23, 28], "xdclassiffierconfus": 10, "xdclassiffierroc": 10, "xg_clf": 10, "xgb": 10, "xgbclassifi": 10, "xgboost": 9, "xgboot": 10, "xgbregressor": 10, "xgparam": 10, "xgtree": 10, "xi": [8, 13, 31], "xi_": 8, "xi_1": 8, "xi_i": 8, "xk": 8, "xla": [13, 21, 28], "xlabel": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 25, 28, 29, 30, 31, 32], "xlim": [6, 10, 32], "xm": 9, "xmesh": 13, "xnew": [0, 13, 28, 30, 31], "xp": 25, "xpanda": [0, 29], "xpd": [5, 11, 29], "xplot": 0, "xscale": [0, 29], "xsr": 9, "xt_x": [13, 30, 31], "xtest": [6, 32], "xtick": [3, 6, 8, 9, 32], "xtrain": [6, 32], "xu": [0, 28], "xx": [0, 22, 28], "xy": [0, 6, 8, 22, 28], "xytext": 8, "xyz": [], "xz": [22, 28], "y": [0, 1, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "y1": 4, "y2": 4, "y3": 4, "y_": [0, 1, 5, 6, 10, 11, 22, 28, 29, 32], "y_0": [0, 5, 11, 22, 28, 29, 32], "y_1": [0, 5, 8, 9, 11, 13, 22, 28, 29, 30, 31, 32], "y_1y_1": 8, "y_1y_1k": 8, "y_1y_2": 8, "y_1y_2k": 8, "y_1y_n": 8, "y_1y_nk": 8, "y_2": [0, 5, 8, 9, 11, 22, 28, 29], "y_2y_1": 8, "y_2y_1k": 8, "y_2y_2": 8, "y_2y_2k": 8, "y_3": [0, 9, 22], "y_4": 22, "y_center": [18, 31], "y_data": [0, 1, 5, 6, 28, 29, 30, 31], "y_data_ful": 1, "y_decis": 8, "y_fit": [0, 29], "y_i": [0, 1, 5, 6, 7, 8, 9, 10, 11, 12, 13, 19, 22, 23, 28, 29, 30, 31, 32], "y_if_": 10, "y_ix_": [0, 28], "y_ix_i": [7, 8, 13, 29, 30, 31], "y_iy_jk": 8, "y_j": [6, 8, 12, 23, 32], "y_k": 12, "y_m": 22, "y_mean": [18, 31], "y_model": [0, 4, 5, 6, 28, 29, 30, 31], "y_n": [8, 13, 30, 31], "y_ny_1": 8, "y_ny_1k": 8, "y_ny_2": 8, "y_ny_2k": 8, "y_ny_n": 8, "y_ny_nk": 8, "y_offset": [6, 17, 29, 31], "y_plot": 9, "y_pred": [0, 1, 4, 6, 7, 8, 9, 10, 29, 31, 32], "y_pred1": 9, "y_pred2": 9, "y_pred_rf": 10, "y_pred_tre": 10, "y_proba": [7, 10], "y_sampl": [], "y_scaler": [6, 29, 31], "y_test": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 29, 30, 31, 32], "y_test_onehot": 1, "y_test_predict": [], "y_tot": 4, "y_train": [0, 1, 3, 4, 5, 6, 7, 9, 10, 11, 15, 16, 17, 19, 28, 29, 30, 31, 32], "y_train_mean": [6, 29, 31], "y_train_onehot": 1, "y_train_predict": [], "y_train_r": 19, "y_train_scal": [6, 29, 31], "y_val": 1, "ye": [3, 6, 7, 32], "year": [0, 21, 28], "yet": [0, 1, 6, 8, 11, 13, 20, 28], "yi": [13, 31], "yield": [0, 2, 5, 6, 8, 10, 12, 13, 14, 22, 25, 28, 30, 31, 32], "yk": 8, "ylabel": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 13, 25, 28, 29, 30, 31, 32], "ylim": [3, 6, 32], "ym": 9, "ymesh": 13, "yn": 0, "yo": [8, 9, 10], "yoshiki": [], "yoshua": [1, 27], "you": [0, 1, 3, 4, 5, 6, 8, 9, 10, 11, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 27, 28, 29, 30, 31, 32], "young": 0, "your": [1, 2, 4, 5, 6, 8, 11, 13, 15, 17, 19, 20, 21, 22, 28, 30, 31, 32], "your_model_object": 16, "yourself": [11, 13, 28, 30], "youtu": [29, 30, 32], "youtub": [21, 30, 31, 32], "ypred": [6, 32], "ypredict": [0, 13, 28, 29, 30, 31], "ypredict2": [13, 30, 31], "ypredictlasso": [5, 30], "ypredictol": [0, 5, 30], "ypredictown": [6, 29, 31], "ypredictownridg": [6, 29, 30, 31], "ypredictridg": [0, 5, 6, 29, 30, 31], "ypredictskl": [6, 29, 31], "ytest": [6, 32], "ytick": [3, 6, 8, 9, 32], "ytild": [0, 6, 28, 29, 32], "ytildelasso": [5, 30], "ytildenp": [0, 28, 29], "ytildeol": [0, 5, 30], "ytildeownridg": [6, 29, 30, 31], "ytilderidg": [5, 6, 29, 30, 31], "ytrain": [6, 32], "yuxi": 28, "yx": [22, 28], "yy": [22, 28], "yz": [22, 28], "z": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 11, 12, 13, 22, 25, 28, 29, 32], "z_": [1, 2, 12, 22, 28], "z_0": [22, 28], "z_1": [22, 28], "z_2": [22, 28], "z_c": 1, "z_h": 1, "z_hidden": 2, "z_i": [1, 12], "z_j": [1, 12], "z_k": [12, 29], "z_m": 1, "z_mod": 9, "z_o": 1, "z_output": 2, "za": [], "zaman": 25, "zaxi": 6, "zero": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 17, 18, 19, 22, 23, 25, 28, 29, 30, 31, 32], "zeros_lik": 4, "zeroth": 29, "zfill": 4, "zip": [4, 6], "zm_h": [0, 28], "zn": [], "zone": [], "zoom": 28, "zscout": [], "zx": [22, 28], "zy": [22, 28], "zz": [22, 28], "\u00f8yvind": [6, 29, 31]}, "titles": ["3. Linear Regression", "14. Building a Feed Forward Neural Network", "15. Solving Differential Equations with Deep Learning", "16. Convolutional Neural Networks", "17. Recurrent neural networks: Overarching view", "4. Ridge and Lasso Regression", "5. Resampling Methods", "6. Logistic Regression", "8. Support Vector Machines, overarching aims", "9. Decision trees, overarching aims", "10. Ensemble Methods: From a Single Tree to Many Trees and Extreme Boosting, Meet the Jungle of Methods", "11. Basic ideas of the Principal Component Analysis (PCA)", "13. Neural networks", "7. Optimization, the central part of any Machine Learning algortithm", "12. Clustering and Unsupervised Learning", "Exercises week 34", "Exercises week 35", "Exercises week 36", "Exercises week 37", "Exercises week 38", "Exercises week 39", "Applied Data Analysis and Machine Learning", "2. Linear Algebra, Handling of Arrays and more Python Features", "Project 1 on Machine Learning, deadline October 6 (midnight), 2025", "Course setting", "1. Elements of Probability Theory and Statistical Data Analysis", "Teachers and Grading", "Textbooks", "Week 34: Introduction to the course, Logistics and Practicalities", "Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression", "Week 36: Linear Regression and Gradient descent", "Week 37: Gradient descent methods", "Week 38: Statistical analysis, bias-variance tradeoff and resampling methods"], "titleterms": {"": [8, 10, 30, 31, 32], "0": [], "04": [], "05": [], "06": [], "07": [], "1": [0, 15, 16, 17, 18, 19, 20, 23, 29], "11": [], "15": [19, 32], "19": 19, "1a": 18, "2": [0, 15, 16, 17, 18, 19, 20, 28, 29, 30], "20": [], "2017": [], "2018": [], "2019": [], "2023": 26, "2025": 23, "27": [], "2a": [], "2b": [], "3": [0, 15, 16, 17, 18, 19, 20, 29], "34": [15, 28], "35": [16, 29], "36": [17, 30], "37": [18, 31], "38": [19, 32], "39": 20, "3a": 18, "3b": 18, "4": [0, 15, 16, 17, 18, 19, 20, 29], "4a": 18, "4b": 18, "5": [0, 16, 18, 19, 20], "6": 23, "8": 31, "A": [0, 1, 4, 8, 9, 28, 32], "And": [28, 29, 31], "But": 31, "In": 26, "Ising": 6, "The": [0, 1, 2, 3, 5, 6, 7, 8, 9, 11, 12, 15, 21, 28, 29, 30, 31, 32], "To": 28, "With": [4, 30], "a11i": [], "about": [28, 29, 30], "abov": 30, "abstract": 20, "accuraci": 31, "across": 31, "activ": [1, 12], "ad": [0, 6, 20, 23, 28, 29], "adaboost": 10, "adagrad": [13, 31], "adam": [13, 31], "adapt": [10, 31], "add": [], "adjust": 1, "advanc": 23, "adversari": 4, "again": [3, 9], "ai": [23, 28], "aim": [8, 9, 28], "aka": 28, "al": 31, "algebra": [22, 28], "algorithm": [9, 10, 11, 12, 28, 29, 30, 31], "algortithm": [13, 30], "all": 8, "an": [0, 4, 10, 15, 20, 28], "analys": [5, 29, 30], "analysi": [0, 5, 6, 11, 21, 23, 25, 28, 29, 30, 32], "analyt": [0, 16, 18], "ani": [13, 30], "anoth": [9, 30, 32], "api": [], "appli": 21, "approach": [0, 8, 14, 28, 31, 32], "approxim": 12, "architectur": 1, "arrai": [22, 28], "assist": 26, "assumpt": 32, "august": [], "author": [], "autocorrel": 25, "autograd": [2, 13, 31], "automat": [13, 31], "avail": 20, "avali": [], "averag": 31, "b": 23, "back": [1, 11, 12, 29, 30], "background": [21, 23, 32], "bag": 10, "base": [13, 31, 32], "basic": [0, 5, 7, 9, 10, 11, 22, 29, 30], "batch": [1, 31], "bay": 5, "befor": 11, "beta": [], "better": 8, "bia": [6, 19, 23, 31, 32], "binari": 1, "bind": 28, "bird": 10, "blind": [], "block": [], "boldsymbol": [18, 29, 32], "book": 19, "boost": 10, "bootstrap": [6, 10, 32], "boston": [], "breast": 1, "brief": [28, 32], "bring": 12, "browser": [], "bsd": [], "build": [1, 3, 9], "c": [23, 28], "calcul": [18, 29, 30], "can": [28, 31, 32], "cancer": [1, 7, 9, 11], "cart": 9, "case": [8, 10, 25, 29, 30, 31], "cdn": [], "cell": [], "central": [13, 21, 25, 30, 32], "chain": 12, "challeng": 31, "chang": 10, "changelog": [], "channel": 28, "chi": [0, 28], "choic": 17, "choos": [1, 31], "cifar01": 3, "citat": [], "classic": 11, "classif": [1, 9, 10], "classifi": 8, "claus": [], "clip": 1, "cluster": 14, "cnn": 3, "code": [1, 2, 5, 9, 11, 12, 13, 14, 15, 16, 20, 23, 28, 29, 30, 31, 32], "collect": [1, 3], "color": [], "colorblind": [], "combin": 31, "commun": 28, "compar": [2, 10, 16], "comparison": [30, 31], "compet": 31, "compil": [], "complet": 29, "complex": [0, 6, 23, 29], "complic": 6, "compon": 11, "comput": [9, 19, 31], "computation": 32, "computerlab": 28, "con": [9, 31], "concept": 25, "condit": 30, "confid": 32, "conjug": 13, "constraint": 31, "contain": [], "content": [], "contn": 28, "contrast": [], "contributor": [], "converg": 31, "convex": [8, 13, 30, 31], "convolut": [3, 12], "copyright": [], "core": [], "correct": 31, "correl": [11, 29], "correspond": [], "cost": [1, 10, 29, 30, 31, 32], "cours": [21, 24, 27, 28], "covari": [5, 11, 25, 29], "cover": 28, "creat": [16, 20], "creator": [], "cross": [6, 23, 32], "cython": 28, "d": 23, "dark": [], "data": [0, 1, 3, 6, 7, 9, 11, 15, 17, 18, 21, 25, 28, 29], "dataset": [1, 3, 18], "david": 28, "deadlin": [23, 28], "deadllin": 26, "decai": [2, 31], "decis": [9, 10], "decomposit": [5, 11, 22, 29, 30], "deeep": [], "deep": [1, 2, 28, 31], "defin": [1, 28], "definit": 19, "deflist": [], "degre": [0, 17, 29], "deliver": [15, 16, 19, 20], "deliveri": 23, "delta": 32, "dens": 0, "depend": [], "deriv": [5, 12, 16, 17, 19, 29, 30, 31, 32], "descent": [2, 10, 13, 18, 23, 30, 31], "design": 29, "detail": [3, 28], "develop": 1, "diagon": 11, "differ": [8, 31], "differenti": [2, 13, 31], "diffus": 2, "dimens": 31, "dimension": [2, 3, 8, 18], "direct": [], "disadvantag": 9, "discret": 25, "discrimin": 28, "distribut": [5, 25, 32], "do": [1, 31], "document": 20, "doe": [29, 30], "domain": 25, "down": 1, "dropout": 1, "e": 23, "economi": [29, 30], "electron": 23, "element": [0, 25, 28], "elimin": 22, "empir": 31, "energi": 28, "ensembl": 10, "entropi": 9, "environ": [0, 15], "equat": [0, 2, 12, 29, 30], "error": [0, 10, 28, 29, 30, 32], "essenti": 28, "estim": 32, "et": 31, "etc": 28, "euler": 2, "evalu": 1, "evid": 31, "exampl": [1, 2, 3, 4, 6, 7, 8, 9, 10, 28, 29, 30, 31, 32], "exercis": [0, 6, 15, 16, 17, 18, 19, 20, 29], "expect": [19, 25, 32], "expens": 32, "experi": 25, "explor": 0, "exponenti": [2, 31], "express": [16, 17, 19, 29], "extend": 30, "extrapol": 4, "extrem": [10, 28], "ey": 10, "f": 23, "fall": 26, "famili": [1, 28], "famou": 22, "fantast": [29, 30], "faq": [], "featur": [9, 16, 22, 29], "februari": [], "feed": [1, 12], "figur": 20, "file": [], "fill": [], "final": [12, 29, 31], "find": [16, 18, 32], "fine": 1, "first": [4, 12, 28, 30], "fit": [0, 10, 15, 16, 28, 30], "fix": [29, 30, 31], "fold": 32, "forc": 3, "forest": 10, "form": 18, "format": [23, 28], "formula": 18, "forward": [1, 2, 12], "foster": 28, "fourier": 3, "frank": 6, "freedom": [0, 17, 29], "frequent": [29, 31], "frequentist": [0, 28], "from": [5, 10, 12, 28, 29, 30, 31, 32], "full": [2, 31], "function": [0, 1, 6, 7, 8, 10, 11, 12, 13, 23, 25, 28, 29, 30, 31, 32], "further": [3, 5, 29, 30], "g": 23, "gan": 4, "gaussian": 22, "gd": [13, 31], "gener": [4, 9, 28], "geometr": [11, 30], "get": 20, "gini": 9, "github": 15, "goal": [15, 16, 17, 18, 19, 20], "good": [0, 20, 28], "goodfellow": 31, "gotthard": [], "grade": [26, 28], "gradient": [1, 2, 10, 13, 18, 23, 30, 31], "greativ": [], "growth": 2, "guid": [], "h": 23, "ha": 21, "handl": [22, 28], "happen": 32, "hessian": [29, 30, 31], "hidden": 2, "high": [], "histogram": 32, "histori": [], "hous": [], "how": 16, "hyperparamet": [1, 17], "hyperplan": 8, "i": [0, 1, 28], "id3": 9, "idea": 11, "ideal": 30, "ident": 32, "identifi": 32, "ii": 28, "iid": 32, "illustr": 30, "implement": [1, 16, 17, 18], "implic": [5, 29, 30], "import": [5, 22, 28, 29, 30], "improv": [1, 31], "includ": [13, 23, 31], "incorpor": [], "increment": 11, "independ": 32, "index": 9, "inform": 26, "input": 2, "instal": [21, 23, 28], "instructor": 26, "interpret": [5, 11, 19, 28, 29, 30, 32], "interv": 32, "introduc": [11, 13, 29], "introduct": [0, 6, 20, 21, 22, 23, 28], "invers": [5, 22], "invert": [29, 30], "ipython": [], "iter": 10, "its": 29, "j": [], "jacobian": 29, "januari": [], "jax": 13, "julia": 28, "jungl": 10, "jupyt": [], "k": 32, "kera": [1, 3], "kernel": [8, 11], "lab": [30, 31, 32], "lagrangian": 8, "lasso": [5, 6, 23, 29, 30], "last": [29, 31], "later": [5, 29, 30], "layer": [1, 2, 3, 12], "learn": [0, 1, 2, 11, 13, 14, 15, 16, 17, 18, 19, 20, 21, 23, 28, 29, 30, 31, 32], "least": [5, 6, 16, 19, 23, 28, 29, 30, 31], "lectur": [28, 30, 31, 32], "level": 10, "librari": [21, 28], "licens": [], "light": [], "likelihood": [7, 32], "limit": [1, 13, 25, 30, 31, 32], "linear": [0, 8, 13, 15, 22, 28, 29, 30], "link": [5, 11, 27, 29, 32], "literatur": 23, "logist": [7, 28], "loss": [29, 30, 31], "lu": 22, "ma": [], "machin": [0, 8, 13, 21, 23, 28, 30], "made": 32, "main": [25, 28], "make": [0, 9, 10, 20, 29], "mani": [10, 12], "markdown": [], "mask": [], "maskedarrai": [], "mass": 28, "materi": [23, 28, 29, 30, 31, 32], "math": [5, 29, 30], "mathemat": [3, 5, 8, 29, 30], "matplotlib": [], "matric": [5, 22, 28], "matrix": [1, 5, 11, 12, 16, 22, 28, 29, 30, 31], "matter": 0, "max": 29, "maximum": 32, "me": [], "mean": [0, 29, 30], "meet": [5, 10, 25, 28, 29], "memori": 31, "mercer": 8, "metadata": [], "method": [6, 9, 10, 13, 23, 28, 30, 31, 32], "metric": 19, "midnight": 23, "min": 29, "mini": 31, "minibatch": 31, "minim": 28, "mit": [], "ml": 28, "mle": 32, "mlp": 12, "mnist": [3, 4], "model": [0, 1, 4, 6, 12, 15, 17, 28], "moment": 31, "momentum": [13, 23, 31], "mondai": [30, 31, 32], "moon": [8, 9], "more": [3, 6, 22, 23, 28, 29, 30, 31, 32], "motiv": 31, "move": 31, "multilay": 12, "multipl": [1, 3, 17], "multipli": 8, "myst": [], "ncsa": [], "need": [23, 28], "network": [1, 2, 3, 4, 7, 12, 28, 31], "neural": [1, 2, 3, 4, 7, 12, 28, 31], "new": [4, 18, 32], "newton": [30, 31], "node": [], "non": [8, 31], "none": 31, "normal": [0, 1, 32], "notat": 12, "note": [23, 29, 30], "notebook": [], "novemb": [], "now": [1, 9, 13, 30, 31, 32], "nuclear": [0, 28], "numba": 28, "number": [0, 2, 25, 29, 31], "numer": [2, 23, 25], "numpi": [22, 28], "object": 3, "obtain": 11, "octob": 23, "od": 2, "off": [6, 19, 23], "ol": [5, 6, 15, 16, 18, 23, 30, 32], "one": [2, 12, 18, 30], "open": [], "oper": 22, "optim": [1, 8, 13, 18, 21, 28, 29, 30, 31], "order": [13, 18, 31], "ordinari": [5, 6, 16, 19, 23, 28, 29, 30, 31], "organ": [0, 28], "oslo": 27, "other": [4, 9, 11, 12, 22, 23, 28], "our": [0, 4, 5, 11, 13, 23, 28, 29, 30], "outcom": [21, 28], "output": 2, "overarch": [0, 4, 8, 9, 28, 29], "overview": [10, 28, 31], "own": [0, 10, 11, 23, 28, 29], "packag": [22, 28], "panda": [28, 29], "paramet": [28, 29], "paramt": 18, "part": [13, 21, 23, 30], "partial": 2, "pass": 1, "pca": 11, "pdf": 25, "perceptron": 12, "perform": [1, 9], "period": 3, "perspect": 1, "pitaya": [], "plan": [29, 30, 31, 32], "plethora": 28, "plot": 32, "point": 4, "poisson": 2, "polici": [], "polynomi": [3, 16, 18, 30], "popul": 2, "popular": 28, "practic": [13, 26, 28, 31], "pre": [1, 3], "preambl": 23, "predict": 4, "preprocess": [29, 31], "prerequisit": [3, 21, 28], "present": 20, "princip": 11, "principl": 3, "pro": [9, 31], "probabl": [5, 25, 32], "problem": [1, 2, 13, 28, 29, 30, 31], "procedur": [9, 28], "process": [1, 3], "program": [2, 13, 23, 30, 31], "project": [6, 20, 23, 26, 28], "prop": 13, "propag": [1, 12], "properti": [5, 25, 29, 30, 31], "python": [0, 9, 15, 21, 22, 28], "quick": 8, "quickli": [], "r": 28, "random": [10, 11, 25], "raphson": 30, "rate": [23, 31], "read": [9, 28, 29, 31, 32], "real": [6, 28], "recommend": [28, 29], "record": [], "recurr": [4, 12], "reduc": [0, 29], "reduct": 3, "refer": 23, "referenc": 20, "reformul": 2, "regress": [0, 5, 6, 7, 9, 10, 13, 15, 17, 18, 19, 23, 28, 29, 30, 31, 32], "regular": 1, "relat": [], "relev": [27, 29], "relu": 1, "remark": 3, "remind": [6, 8, 28, 29, 30, 31], "replac": [13, 31], "report": [20, 23], "repositori": [15, 32], "requir": [2, 21], "resampl": [6, 19, 23, 32], "rescal": [6, 29], "residu": [29, 30], "resourc": 2, "result": [29, 30], "revis": [], "revisit": [13, 30, 31], "rewrit": [28, 29, 32], "ridg": [0, 5, 6, 17, 18, 19, 23, 29, 30, 31], "rm": 13, "rmsprop": 31, "role": [], "rule": [12, 31], "rung": 23, "same": [13, 31, 32], "sampl": 11, "scalabl": 31, "scale": [17, 18, 19, 29, 31], "schedul": 28, "schemat": 9, "scheme": 2, "scienc": 28, "scikit": [0, 1, 11, 28, 29, 30, 31, 32], "second": [13, 18, 31], "semest": 26, "sensit": 30, "septemb": [19, 30, 31, 32], "session": [30, 31, 32], "set": [0, 2, 3, 9, 12, 15, 24, 28, 29, 30], "setup": 15, "sgd": [13, 31], "should": 1, "show": [], "similar": [13, 31], "simpl": [0, 4, 9, 13, 18, 28, 29, 30, 31], "simplest": 18, "singl": 10, "singular": [5, 11, 29, 30], "size": [29, 30, 31], "sklearn": 16, "slightli": 31, "smoothi": [], "sneak": 31, "soft": 8, "softmax": 1, "softwar": [23, 28], "solv": [2, 30], "solver": 13, "some": [13, 22, 29, 30], "sourc": [], "specifi": 2, "speed": 31, "sphinx": [], "split": [0, 15, 29], "squar": [0, 5, 6, 10, 16, 19, 23, 28, 29, 30, 31], "standard": [13, 29, 32], "start": 20, "state": 0, "statist": [5, 6, 21, 25, 28, 32], "steepest": [10, 13, 30], "step": [31, 32], "stochast": [13, 23, 25, 31], "stop": 31, "strongli": [28, 31], "structur": [], "suggest": 28, "sum": 32, "summari": [26, 28], "superposit": 3, "supervis": 1, "support": 8, "svd": [5, 29, 30], "synthet": 18, "systemat": 3, "t": 29, "take": 16, "taken": [28, 31], "teach": 26, "teacher": [26, 28], "team": [], "technic": 29, "techniqu": [6, 11, 23], "technologi": 21, "tensorflow": [1, 3], "tent": [26, 28], "term": 32, "test": [0, 1, 15, 17, 29], "texmath": [], "text": 28, "textbook": [27, 28], "than": 30, "thank": [], "theorem": [5, 8, 11, 12, 25, 32], "theoret": 31, "theori": 25, "theta": [18, 32], "thi": 28, "time": 31, "tip": [13, 31], "todo": [], "togeth": 12, "tool": [23, 28], "top": 1, "topic": 28, "toward": 11, "trade": [6, 19, 23], "tradeoff": [6, 32], "train": [0, 1, 4, 15, 28, 29], "transform": 3, "translat": [], "tree": [9, 10], "tuesdai": 30, "tune": 1, "two": [3, 8, 21], "type": [2, 4, 12, 28], "uio": 28, "understand": 32, "univers": [12, 27], "unsupervis": 14, "up": [0, 2, 9, 12, 15, 28, 29, 30, 32], "updat": [23, 31], "us": [0, 1, 2, 3, 7, 13, 16, 18, 19, 21, 23, 28, 29, 30, 31], "usag": 31, "v": [3, 31], "valid": [6, 23, 32], "valu": [5, 11, 19, 25, 29, 30, 32], "vari": 31, "variabl": [25, 30], "varianc": [6, 19, 23, 32], "variou": [0, 32], "vector": [8, 12, 16, 22, 28, 29], "versu": 28, "video": [31, 32], "view": [0, 4, 10, 29], "virtual": 15, "visual": [1, 9], "wai": [9, 23, 32], "wave": 2, "we": [28, 31], "wednesdai": 30, "week": [15, 16, 17, 18, 19, 20, 28, 29, 30, 31, 32], "weekli": [], "welcom": [], "what": [0, 28, 29, 30, 32], "when": 31, "which": [1, 31], "why": [28, 31, 32], "wisconsin": 7, "workflow": [], "wrap": 32, "write": [4, 11, 20, 23, 30], "x": 29, "xgboost": 10, "yaml": [], "yet": 30, "your": [0, 10, 16, 18, 23, 29]}}) \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/week38.html b/doc/LectureNotes/_build/html/week38.html index 9096b83ea..183739715 100644 --- a/doc/LectureNotes/_build/html/week38.html +++ b/doc/LectureNotes/_build/html/week38.html @@ -391,7 +391,7 @@ document.write(`
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \(\boldsymbol{\beta}\)
  • +
  • Expectation value and variance for \(\boldsymbol{\theta}\)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -474,7 +474,7 @@ move from a linear algebra analysis to a statistical analysis. In particular, we will focus on what the regularization terms can result in. We will amongst other things show that the regularization parameter can reduce considerably the variance of the parameters -\(\beta\).

    +\(\theta\).

    On of the advantages of doing linear regression is that we actually end up with analytical expressions for several statistical quantities.
    Standard least squares and Ridge regression allow us to @@ -494,7 +494,7 @@ independent, i.e.:

    The randomness of \(\varepsilon_i\) implies that \(\mathbf{y}_i\) is also a random variable. In particular, \(\mathbf{y}_i\) is normally distributed, because \(\varepsilon_i \sim -\mathcal{N}(0, \sigma^2)\) and \(\mathbf{X}_{i,\ast} \, \boldsymbol{\beta}\) is a +\mathcal{N}(0, \sigma^2)\) and \(\mathbf{X}_{i,\ast} \, \boldsymbol{\theta}\) is a non-random scalar. To specify the parameters of the distribution of \(\mathbf{y}_i\) we need to calculate its first two moments.

    Recall that \(\boldsymbol{X}\) is a matrix of dimensionality \(n\times p\). The @@ -514,7 +514,7 @@ which describe our data

    function \(f\) is approximated by \(\boldsymbol{\tilde{y}}\) where we want to minimize \((\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\), our MSE, with

    \[ -\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\theta}. \]
    @@ -524,8 +524,8 @@ function \(f\) is approximated \[ \begin{align*} \mathbb{E}(y_i) & = -\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) -\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\theta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \theta, \end{align*} \]

    while @@ -535,83 +535,83 @@ its variance is

    \begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i - \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - [\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, -\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & -= \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i -\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, -\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 -\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + -\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\theta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \theta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, \mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. \end{align*} \end{split}\] -

    Hence, \(y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2)\), that is \(\boldsymbol{y}\) follows a normal distribution with -mean value \(\boldsymbol{X}\boldsymbol{\beta}\) and variance \(\sigma^2\) (not be confused with the singular values of the SVD).

    +

    Hence, \(y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta}, \sigma^2)\), that is \(\boldsymbol{y}\) follows a normal distribution with +mean value \(\boldsymbol{X}\boldsymbol{\theta}\) and variance \(\sigma^2\) (not be confused with the singular values of the SVD).

    -
    -

    Expectation value and variance for \(\boldsymbol{\beta}\)#

    -

    With the OLS expressions for the optimal parameters \(\boldsymbol{\hat{\beta}}\) we can evaluate the expectation value

    +
    +

    Expectation value and variance for \(\boldsymbol{\theta}\)#

    +

    With the OLS expressions for the optimal parameters \(\boldsymbol{\hat{\theta}}\) we can evaluate the expectation value

    \[ -\mathbb{E}(\boldsymbol{\hat{\beta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +\mathbb{E}(\boldsymbol{\hat{\theta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\theta}=\boldsymbol{\theta}. \]

    This means that the estimator of the regression parameters is unbiased.

    We can also calculate the variance

    -

    The variance of the optimal value \(\boldsymbol{\hat{\beta}}\) is

    +

    The variance of the optimal value \(\boldsymbol{\hat{\theta}}\) is

    \[\begin{split} \begin{eqnarray*} -\mbox{Var}(\boldsymbol{\hat{\beta}}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\mbox{Var}(\boldsymbol{\hat{\theta}}) & = & \mathbb{E} \{ [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})] [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})]^{T} \} \\ -& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}]^{T} \} \\ -% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} % \\ -% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\theta} \boldsymbol{\theta}^T \\ -& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, \end{eqnarray*} \end{split}\]

    where we have used that \(\mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = -\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + -\sigma^2 \, \mathbf{I}_{nn}\). From \(\mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn}\). From \(\mbox{Var}(\boldsymbol{\theta}) = \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}\), one obtains an estimate of the variance of the estimate of the \(j\)-th regression coefficient: -\(\boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to +\(\boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to construct a confidence interval for the estimates.

    In a similar way, we can obtain analytical expressions for say the -expectation values of the parameters \(\boldsymbol{\beta}\) and their variance +expectation values of the parameters \(\boldsymbol{\theta}\) and their variance when we employ Ridge regression, allowing us again to define a confidence interval.

    It is rather straightforward to show that

    \[ -\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +\mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\theta}^{\mathrm{OLS}}. \]

    We see clearly that -\(\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}}\) for any \(\lambda > 0\). We say then that the ridge estimator is biased.

    +\(\mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\theta}^{\mathrm{OLS}}\) for any \(\lambda > 0\). We say then that the ridge estimator is biased.

    We can also compute the variance as

    \[ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, \]
    -

    and it is easy to see that if the parameter \(\lambda\) goes to infinity then the variance of Ridge parameters \(\boldsymbol{\beta}\) goes to zero.

    +

    and it is easy to see that if the parameter \(\lambda\) goes to infinity then the variance of Ridge parameters \(\boldsymbol{\theta}\) goes to zero.

    With this, we can compute the difference

    \[ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\theta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. \]

    The difference is non-negative definite since each component of the matrix product is non-negative definite. -This means the variance we obtain with the standard OLS will always for \(\lambda > 0\) be larger than the variance of \(\boldsymbol{\beta}\) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.

    +This means the variance we obtain with the standard OLS will always for \(\lambda > 0\) be larger than the variance of \(\boldsymbol{\theta}\) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.

    Deriving OLS from a probability distribution#

    @@ -621,14 +621,14 @@ that our output is determined by a given continuous function distribution with zero mean value and an undetermined variance \(\sigma^2\).

    We found above that the outputs \(\boldsymbol{y}\) have a mean value given by -\(\boldsymbol{X}\hat{\boldsymbol{\beta}}\) and variance \(\sigma^2\). Since the entries to +\(\boldsymbol{X}\hat{\boldsymbol{\theta}}\) and variance \(\sigma^2\). Since the entries to the design matrix are not stochastic variables, we can assume that the probability distribution of our targets is also a normal distribution -but now with mean value \(\boldsymbol{X}\hat{\boldsymbol{\beta}}\). This means that a +but now with mean value \(\boldsymbol{X}\hat{\boldsymbol{\theta}}\). This means that a single output \(y_i\) is given by the Gaussian distribution

    \[ -y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\beta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\theta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. \]
    @@ -637,13 +637,13 @@ y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\beta}, \sigma^2)=\frac{1}{\ We define this distribution as

    \[ -p(y_i, \boldsymbol{X}\vert\boldsymbol{\beta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}, +p(y_i, \boldsymbol{X}\vert\boldsymbol{\theta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}, \]
    -

    which reads as finding the likelihood of an event \(y_i\) with the input variables \(\boldsymbol{X}\) given the parameters (to be determined) \(\boldsymbol{\beta}\).

    +

    which reads as finding the likelihood of an event \(y_i\) with the input variables \(\boldsymbol{X}\) given the parameters (to be determined) \(\boldsymbol{\theta}\).

    Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event \(\boldsymbol{y}\) as the product of the single events, that is we have

    \[ -p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta}). +p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta}). \]

    We will write this in a more compact form reserving \(\boldsymbol{D}\) for the domain of events, including the ouputs (targets) and the inputs. That is in case we have a simple one-dimensional input and output case

    @@ -655,9 +655,9 @@ in case we have a simple one-dimensional input and output case

    We can now rewrite the above probability as

    \[ -p(\boldsymbol{D}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +p(\boldsymbol{D}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. \]
    -

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \(\boldsymbol{D}\) given a set of parameters \(\boldsymbol{\beta}\).

    +

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \(\boldsymbol{D}\) given a set of parameters \(\boldsymbol{\theta}\).

    Maximum Likelihood Estimation (MLE)#

    @@ -667,7 +667,7 @@ given some observed data. This is achieved by maximizing a likelihood function so that, under the assumed statistical model, the observed data is the most probable.

    We will assume here that our events are given by the above Gaussian -distribution and we will determine the optimal parameters \(\beta\) by +distribution and we will determine the optimal parameters \(\theta\) by maximizing the above PDF. However, computing the derivatives of a product function is cumbersome and can easily lead to overflow and/or underflowproblems, with potentials for loss of numerical precision.

    @@ -684,22 +684,22 @@ is equivalent to the maximization/minimization of the function itself.

    We could now define a new cost function to minimize, namely the negative logarithm of the above PDF

    \[ -C(\boldsymbol{\beta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}, +C(\boldsymbol{\theta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}, \]

    which becomes

    \[ -C(\boldsymbol{\beta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\vert\vert_2^2}{2\sigma^2}. +C(\boldsymbol{\theta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})\vert\vert_2^2}{2\sigma^2}. \]
    -

    Taking the derivative of the new cost function with respect to the parameters \(\beta\) we recognize our familiar OLS equation, namely

    +

    Taking the derivative of the new cost function with respect to the parameters \(\theta\) we recognize our familiar OLS equation, namely

    \[ -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right) =0, +\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta}\right) =0, \]
    -

    which leads to the well-known OLS equation for the optimal paramters \(\beta\)

    +

    which leads to the well-known OLS equation for the optimal paramters \(\theta\)

    \[ -\hat{\boldsymbol{\beta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! +\hat{\boldsymbol{\theta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! \]

    Next week we will make a similar analysis for Ridge and Lasso regression

    @@ -926,23 +926,23 @@ finite \(m\), it is not always

    Confidence Intervals#

    Confidence intervals are used in statistics and represent a type of estimate computed from the observed data. This gives a range of values for an -unknown parameter such as the parameters \(\boldsymbol{\beta}\) from linear regression.

    -

    With the OLS expressions for the parameters \(\boldsymbol{\beta}\) we found -\(\mathbb{E}(\boldsymbol{\beta}) = \boldsymbol{\beta}\), which means that the estimator of the regression parameters is unbiased.

    +unknown parameter such as the parameters \(\boldsymbol{\theta}\) from linear regression.

    +

    With the OLS expressions for the parameters \(\boldsymbol{\theta}\) we found +\(\mathbb{E}(\boldsymbol{\theta}) = \boldsymbol{\theta}\), which means that the estimator of the regression parameters is unbiased.

    In the exercises this week we show that the variance of the estimate of the \(j\)-th regression coefficient is -\(\boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \).

    +\(\boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \).

    This quantity can be used to construct a confidence interval for the estimates.

    Standard Approach based on the Normal Distribution#

    -

    We will assume that the parameters \(\beta\) follow a normal +

    We will assume that the parameters \(\theta\) follow a normal distribution. We can then define the confidence interval. Here we will be using as -shorthands \(\mu_{\beta}\) for the above mean value and \(\sigma_{\beta}\) +shorthands \(\mu_{\theta}\) for the above mean value and \(\sigma_{\theta}\) for the standard deviation. We have then a confidence interval

    \[ -\left(\mu_{\beta}\pm \frac{z\sigma_{\beta}}{\sqrt{n}}\right), +\left(\mu_{\theta}\pm \frac{z\sigma_{\theta}}{\sqrt{n}}\right), \]

    where \(z\) defines the level of certainty (or confidence). For a normal distribution typical parameters are \(z=2.576\) which corresponds to a @@ -956,11 +956,11 @@ Bootstrap method, why it works and various theorems related to it.

    Resampling methods: Bootstrap background#

    -

    Since \(\widehat{\beta} = \widehat{\beta}(\boldsymbol{X})\) is a function of random variables, -\(\widehat{\beta}\) itself must be a random variable. Thus it has +

    Since \(\widehat{\theta} = \widehat{\theta}(\boldsymbol{X})\) is a function of random variables, +\(\widehat{\theta}\) itself must be a random variable. Thus it has a pdf, call this function \(p(\boldsymbol{t})\). The aim of the bootstrap is to estimate \(p(\boldsymbol{t})\) by the relative frequency of -\(\widehat{\beta}\). You can think of this as using a histogram +\(\widehat{\theta}\). You can think of this as using a histogram in the place of \(p(\boldsymbol{t})\). If the relative frequency closely resembles \(p(\vec{t})\), then using numerics, it is straight forward to estimate all the interesting parameters of \(p(\boldsymbol{t})\) using point @@ -968,18 +968,18 @@ estimators.

    Resampling methods: More Bootstrap background#

    -

    In the case that \(\widehat{\beta}\) has +

    In the case that \(\widehat{\theta}\) has more than one component, and the components are independent, we use the same estimator on each component separately. If the probability density function of \(X_i\), \(p(x)\), had been known, then it would have been straightforward to do this by:

    1. Drawing lots of numbers from \(p(x)\), suppose we call one such set of numbers \((X_1^*, X_2^*, \cdots, X_n^*)\).

    2. -
    3. Then using these numbers, we could compute a replica of \(\widehat{\beta}\) called \(\widehat{\beta}^*\).

    4. +
    5. Then using these numbers, we could compute a replica of \(\widehat{\theta}\) called \(\widehat{\theta}^*\).

    By repeated use of the above two points, many -estimates of \(\widehat{\beta}\) can be obtained. The -idea is to use the relative frequency of \(\widehat{\beta}^*\) +estimates of \(\widehat{\theta}\) can be obtained. The +idea is to use the relative frequency of \(\widehat{\theta}^*\) (think of a histogram) as an estimate of \(p(\boldsymbol{t})\).

    @@ -1000,18 +1000,18 @@ result in some asymptotic sense? The answer is yes.

    1. Draw with replacement \(n\) numbers for the observed variables \(\boldsymbol{x} = (x_1,x_2,\cdots,x_n)\).

    2. Define a vector \(\boldsymbol{x}^*\) containing the values which were drawn from \(\boldsymbol{x}\).

    3. -
    4. Using the vector \(\boldsymbol{x}^*\) compute \(\widehat{\beta}^*\) by evaluating \(\widehat \beta\) under the observations \(\boldsymbol{x}^*\).

    5. +
    6. Using the vector \(\boldsymbol{x}^*\) compute \(\widehat{\theta}^*\) by evaluating \(\widehat \theta\) under the observations \(\boldsymbol{x}^*\).

    7. Repeat this process \(k\) times.

    When you are done, you can draw a histogram of the relative frequency -of \(\widehat \beta^*\). This is your estimate of the probability +of \(\widehat \theta^*\). This is your estimate of the probability distribution \(p(t)\). Using this probability distribution you can estimate any statistics thereof. In principle you never draw the -histogram of the relative frequency of \(\widehat{\beta}^*\). Instead +histogram of the relative frequency of \(\widehat{\theta}^*\). Instead you use the estimators corresponding to the statistic of interest. For example, if you are interested in estimating the variance of \(\widehat -\beta\), apply the etsimator \(\widehat \sigma^2\) to the values -\(\widehat \beta^*\).

    +\theta\), apply the etsimator \(\widehat \sigma^2\) to the values +\(\widehat \theta^*\).

    Code example for the Bootstrap method#

    @@ -1096,12 +1096,12 @@ tasks. Consider a dataset \(\mathcal{

    where \(\epsilon\) is normally distributed with mean zero and standard deviation \(\sigma^2\).

    In our derivation of the ordinary least squares method we defined then an approximation to the function \(f\) in terms of the parameters -\(\boldsymbol{\beta}\) and the design matrix \(\boldsymbol{X}\) which embody our model, -that is \(\boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta}\).

    -

    Thereafter we found the parameters \(\boldsymbol{\beta}\) by optimizing the means squared error via the so-called cost function

    +\(\boldsymbol{\theta}\) and the design matrix \(\boldsymbol{X}\) which embody our model, +that is \(\boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\theta}\).

    +

    Thereafter we found the parameters \(\boldsymbol{\theta}\) by optimizing the means squared error via the so-called cost function

    \[ -C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +C(\boldsymbol{X},\boldsymbol{\theta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. \]

    We can rewrite this as

    @@ -1734,7 +1734,7 @@ the jupyter-notebook from week 37 (September 12-16).

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \(\boldsymbol{\beta}\)
  • +
  • Expectation value and variance for \(\boldsymbol{\theta}\)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/LectureNotes/_build/jupyter_execute/week38.ipynb b/doc/LectureNotes/_build/jupyter_execute/week38.ipynb index eb429e0cc..7dd2e229a 100644 --- a/doc/LectureNotes/_build/jupyter_execute/week38.ipynb +++ b/doc/LectureNotes/_build/jupyter_execute/week38.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "3896fc4b", + "id": "169923b3", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "ec008b77", + "id": "47013ee7", "metadata": { "editable": true }, @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "997c17fc", + "id": "d1fb5464", "metadata": { "editable": true }, @@ -47,7 +47,7 @@ }, { "cell_type": "markdown", - "id": "893cd04d", + "id": "1a34a7ce", "metadata": { "editable": true }, @@ -68,7 +68,7 @@ }, { "cell_type": "markdown", - "id": "115e99b5", + "id": "c6f56f83", "metadata": { "editable": true }, @@ -81,7 +81,7 @@ "particular, we will focus on what the regularization terms can result\n", "in. We will amongst other things show that the regularization\n", "parameter can reduce considerably the variance of the parameters\n", - "$\\beta$.\n", + "$\\theta$.\n", "\n", "On of the advantages of doing linear regression is that we actually end up with\n", "analytical expressions for several statistical quantities. \n", @@ -96,7 +96,7 @@ }, { "cell_type": "markdown", - "id": "857956f2", + "id": "3ece0c04", "metadata": { "editable": true }, @@ -112,7 +112,7 @@ }, { "cell_type": "markdown", - "id": "074e7f9c", + "id": "d36cf6db", "metadata": { "editable": true }, @@ -120,7 +120,7 @@ "The randomness of $\\varepsilon_i$ implies that\n", "$\\mathbf{y}_i$ is also a random variable. In particular,\n", "$\\mathbf{y}_i$ is normally distributed, because $\\varepsilon_i \\sim\n", - "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\beta}$ is a\n", + "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\theta}$ is a\n", "non-random scalar. To specify the parameters of the distribution of\n", "$\\mathbf{y}_i$ we need to calculate its first two moments. \n", "\n", @@ -131,7 +131,7 @@ }, { "cell_type": "markdown", - "id": "0220f12c", + "id": "5903d2be", "metadata": { "editable": true }, @@ -145,7 +145,7 @@ }, { "cell_type": "markdown", - "id": "fe96cf24", + "id": "37f4e199", "metadata": { "editable": true }, @@ -157,7 +157,7 @@ }, { "cell_type": "markdown", - "id": "b308e3f9", + "id": "7a7b084f", "metadata": { "editable": true }, @@ -168,19 +168,19 @@ }, { "cell_type": "markdown", - "id": "a2d28d54", + "id": "1ec511a1", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta}.\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "501243ea", + "id": "ffe76935", "metadata": { "editable": true }, @@ -192,7 +192,7 @@ }, { "cell_type": "markdown", - "id": "a0735fd3", + "id": "e569274a", "metadata": { "editable": true }, @@ -200,15 +200,15 @@ "$$\n", "\\begin{align*} \n", "\\mathbb{E}(y_i) & =\n", - "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}) + \\mathbb{E}(\\varepsilon_i)\n", - "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\beta, \n", + "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}) + \\mathbb{E}(\\varepsilon_i)\n", + "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\theta, \n", "\\end{align*}\n", "$$" ] }, { "cell_type": "markdown", - "id": "105564c8", + "id": "c4dd2623", "metadata": { "editable": true }, @@ -219,7 +219,7 @@ }, { "cell_type": "markdown", - "id": "e484179a", + "id": "df2f8936", "metadata": { "editable": true }, @@ -228,12 +228,12 @@ "\\begin{align*} \\mbox{Var}(y_i) & = \\mathbb{E} \\{ [y_i\n", "- \\mathbb{E}(y_i)]^2 \\} \\, \\, \\, = \\, \\, \\, \\mathbb{E} ( y_i^2 ) -\n", "[\\mathbb{E}(y_i)]^2 \\\\ & = \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\,\n", - "\\beta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \\\\ &\n", - "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2 \\varepsilon_i\n", - "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", - "\\ast} \\, \\beta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2\n", - "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} +\n", - "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \n", + "\\theta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \\\\ &\n", + "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2 \\varepsilon_i\n", + "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", + "\\ast} \\, \\theta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2\n", + "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} +\n", + "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \n", "\\\\ & = \\mathbb{E}(\\varepsilon_i^2 ) \\, \\, \\, = \\, \\, \\,\n", "\\mbox{Var}(\\varepsilon_i) \\, \\, \\, = \\, \\, \\, \\sigma^2. \n", "\\end{align*}\n", @@ -242,42 +242,42 @@ }, { "cell_type": "markdown", - "id": "028be304", + "id": "0a3e5956", "metadata": { "editable": true }, "source": [ - "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", - "mean value $\\boldsymbol{X}\\boldsymbol{\\beta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." + "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", + "mean value $\\boldsymbol{X}\\boldsymbol{\\theta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." ] }, { "cell_type": "markdown", - "id": "71256ba5", + "id": "973e45a3", "metadata": { "editable": true }, "source": [ - "## Expectation value and variance for $\\boldsymbol{\\beta}$\n", + "## Expectation value and variance for $\\boldsymbol{\\theta}$\n", "\n", - "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\beta}}$ we can evaluate the expectation value" + "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\theta}}$ we can evaluate the expectation value" ] }, { "cell_type": "markdown", - "id": "236fd8a6", + "id": "ff486d1e", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E}(\\boldsymbol{\\hat{\\beta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\beta}=\\boldsymbol{\\beta}.\n", + "\\mathbb{E}(\\boldsymbol{\\hat{\\theta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\theta}=\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "8d2e261d", + "id": "e4307815", "metadata": { "editable": true }, @@ -286,35 +286,35 @@ "\n", "We can also calculate the variance\n", "\n", - "The variance of the optimal value $\\boldsymbol{\\hat{\\beta}}$ is" + "The variance of the optimal value $\\boldsymbol{\\hat{\\theta}}$ is" ] }, { "cell_type": "markdown", - "id": "2d8ebdef", + "id": "490b2cbf", "metadata": { "editable": true }, "source": [ "$$\n", "\\begin{eqnarray*}\n", - "\\mbox{Var}(\\boldsymbol{\\hat{\\beta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})] [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})]^{T} \\}\n", + "\\mbox{Var}(\\boldsymbol{\\hat{\\theta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})] [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})]^{T} \\}\n", "\\\\\n", - "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}]^{T} \\}\n", + "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}]^{T} \\}\n", "\\\\\n", - "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", + "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", "% \\\\\n", - "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\boldsymbol{\\beta}^T\n", + "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\boldsymbol{\\theta}^T\n", "\\\\\n", - "& = & \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\, \\, \\, = \\, \\, \\, \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1},\n", "\\end{eqnarray*}\n", "$$" @@ -322,21 +322,21 @@ }, { "cell_type": "markdown", - "id": "3285f57f", + "id": "5d8dd6bc", "metadata": { "editable": true }, "source": [ "where we have used that $\\mathbb{E} (\\mathbf{Y} \\mathbf{Y}^{T}) =\n", - "\\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} +\n", - "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\beta}) = \\sigma^2\n", + "\\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} +\n", + "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\theta}) = \\sigma^2\n", "\\, (\\mathbf{X}^{T} \\mathbf{X})^{-1}$, one obtains an estimate of the\n", "variance of the estimate of the $j$-th regression coefficient:\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", "construct a confidence interval for the estimates.\n", "\n", "In a similar way, we can obtain analytical expressions for say the\n", - "expectation values of the parameters $\\boldsymbol{\\beta}$ and their variance\n", + "expectation values of the parameters $\\boldsymbol{\\theta}$ and their variance\n", "when we employ Ridge regression, allowing us again to define a confidence interval. \n", "\n", "It is rather straightforward to show that" @@ -344,80 +344,80 @@ }, { "cell_type": "markdown", - "id": "925312c4", + "id": "3594078a", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\beta}^{\\mathrm{OLS}}.\n", + "\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\theta}^{\\mathrm{OLS}}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "b9fd6bed", + "id": "43007b7a", "metadata": { "editable": true }, "source": [ "We see clearly that \n", - "$\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\beta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", + "$\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\theta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", "\n", "We can also compute the variance as" ] }, { "cell_type": "markdown", - "id": "d919b998", + "id": "3cd4e7da", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", "$$" ] }, { "cell_type": "markdown", - "id": "a878a50d", + "id": "5816b21e", "metadata": { "editable": true }, "source": [ - "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\beta}$ goes to zero. \n", + "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\theta}$ goes to zero. \n", "\n", "With this, we can compute the difference" ] }, { "cell_type": "markdown", - "id": "ccbbea14", + "id": "b58b8607", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\beta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\theta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "74775fd2", + "id": "c843db7f", "metadata": { "editable": true }, "source": [ "The difference is non-negative definite since each component of the\n", "matrix product is non-negative definite. \n", - "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." + "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\theta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." ] }, { "cell_type": "markdown", - "id": "032b9317", + "id": "e81b1c39", "metadata": { "editable": true }, @@ -431,28 +431,28 @@ "$\\sigma^2$.\n", "\n", "We found above that the outputs $\\boldsymbol{y}$ have a mean value given by\n", - "$\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$ and variance $\\sigma^2$. Since the entries to\n", + "$\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$ and variance $\\sigma^2$. Since the entries to\n", "the design matrix are not stochastic variables, we can assume that the\n", "probability distribution of our targets is also a normal distribution\n", - "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$. This means that a\n", + "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$. This means that a\n", "single output $y_i$ is given by the Gaussian distribution" ] }, { "cell_type": "markdown", - "id": "1c61251d", + "id": "ea9719e3", "metadata": { "editable": true }, "source": [ "$$\n", - "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "5545f5c8", + "id": "c28b598f", "metadata": { "editable": true }, @@ -465,43 +465,43 @@ }, { "cell_type": "markdown", - "id": "470b75b2", + "id": "88063ed7", "metadata": { "editable": true }, "source": [ "$$\n", - "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]},\n", + "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]},\n", "$$" ] }, { "cell_type": "markdown", - "id": "908e209b", + "id": "a34e4c28", "metadata": { "editable": true }, "source": [ - "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\beta}$.\n", + "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\theta}$.\n", "\n", "Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event $\\boldsymbol{y}$ as the product of the single events, that is we have" ] }, { "cell_type": "markdown", - "id": "4fdbf729", + "id": "1fc7ae0a", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta}).\n", + "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta}).\n", "$$" ] }, { "cell_type": "markdown", - "id": "ac5dc52a", + "id": "30d5a1be", "metadata": { "editable": true }, @@ -512,7 +512,7 @@ }, { "cell_type": "markdown", - "id": "dc35e3e1", + "id": "1d29fd29", "metadata": { "editable": true }, @@ -524,7 +524,7 @@ }, { "cell_type": "markdown", - "id": "f0b84ac4", + "id": "289a116c", "metadata": { "editable": true }, @@ -535,29 +535,29 @@ }, { "cell_type": "markdown", - "id": "90eb85ae", + "id": "d1cfbb56", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{D}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "p(\\boldsymbol{D}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "5e70b7b1", + "id": "e089fc0e", "metadata": { "editable": true }, "source": [ - "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\beta}$." + "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\theta}$." ] }, { "cell_type": "markdown", - "id": "dcf0e487", + "id": "32cf9944", "metadata": { "editable": true }, @@ -571,7 +571,7 @@ "data is the most probable. \n", "\n", "We will assume here that our events are given by the above Gaussian\n", - "distribution and we will determine the optimal parameters $\\beta$ by\n", + "distribution and we will determine the optimal parameters $\\theta$ by\n", "maximizing the above PDF. However, computing the derivatives of a\n", "product function is cumbersome and can easily lead to overflow and/or\n", "underflowproblems, with potentials for loss of numerical precision.\n", @@ -588,7 +588,7 @@ }, { "cell_type": "markdown", - "id": "c14bbd71", + "id": "ee8a9544", "metadata": { "editable": true }, @@ -600,19 +600,19 @@ }, { "cell_type": "markdown", - "id": "13fcb784", + "id": "4bfeb203", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})},\n", + "C(\\boldsymbol{\\theta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})},\n", "$$" ] }, { "cell_type": "markdown", - "id": "8f685501", + "id": "256143f9", "metadata": { "editable": true }, @@ -622,63 +622,63 @@ }, { "cell_type": "markdown", - "id": "1b83ae5f", + "id": "60d75bb1", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})\\vert\\vert_2^2}{2\\sigma^2}.\n", + "C(\\boldsymbol{\\theta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta})\\vert\\vert_2^2}{2\\sigma^2}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "c44b7952", + "id": "7b111014", "metadata": { "editable": true }, "source": [ - "Taking the derivative of the *new* cost function with respect to the parameters $\\beta$ we recognize our familiar OLS equation, namely" + "Taking the derivative of the *new* cost function with respect to the parameters $\\theta$ we recognize our familiar OLS equation, namely" ] }, { "cell_type": "markdown", - "id": "7848c18a", + "id": "0f42a0b5", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right) =0,\n", + "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta}\\right) =0,\n", "$$" ] }, { "cell_type": "markdown", - "id": "bd67b48f", + "id": "39a9276b", "metadata": { "editable": true }, "source": [ - "which leads to the well-known OLS equation for the optimal paramters $\\beta$" + "which leads to the well-known OLS equation for the optimal paramters $\\theta$" ] }, { "cell_type": "markdown", - "id": "8a8665d9", + "id": "84c63927", "metadata": { "editable": true }, "source": [ "$$\n", - "\\hat{\\boldsymbol{\\beta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", + "\\hat{\\boldsymbol{\\theta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", "$$" ] }, { "cell_type": "markdown", - "id": "718fe69c", + "id": "c087ef22", "metadata": { "editable": true }, @@ -688,7 +688,7 @@ }, { "cell_type": "markdown", - "id": "f837afb3", + "id": "79503987", "metadata": { "editable": true }, @@ -707,7 +707,7 @@ }, { "cell_type": "markdown", - "id": "82ed1dd7", + "id": "c165025b", "metadata": { "editable": true }, @@ -735,7 +735,7 @@ }, { "cell_type": "markdown", - "id": "06046f8f", + "id": "efb63405", "metadata": { "editable": true }, @@ -761,7 +761,7 @@ }, { "cell_type": "markdown", - "id": "7639ed0d", + "id": "88d4bfd8", "metadata": { "editable": true }, @@ -778,7 +778,7 @@ }, { "cell_type": "markdown", - "id": "dc080604", + "id": "6a0795e0", "metadata": { "editable": true }, @@ -798,7 +798,7 @@ }, { "cell_type": "markdown", - "id": "11aa170a", + "id": "d815fdf3", "metadata": { "editable": true }, @@ -827,7 +827,7 @@ }, { "cell_type": "markdown", - "id": "a56444a1", + "id": "a5b26ed1", "metadata": { "editable": true }, @@ -852,7 +852,7 @@ }, { "cell_type": "markdown", - "id": "020c8c0c", + "id": "e5783b81", "metadata": { "editable": true }, @@ -872,7 +872,7 @@ }, { "cell_type": "markdown", - "id": "f707010c", + "id": "9bdfff4f", "metadata": { "editable": true }, @@ -884,7 +884,7 @@ }, { "cell_type": "markdown", - "id": "062d8d7c", + "id": "bd1e4a83", "metadata": { "editable": true }, @@ -894,7 +894,7 @@ }, { "cell_type": "markdown", - "id": "adf6bd14", + "id": "66d14a83", "metadata": { "editable": true }, @@ -909,7 +909,7 @@ }, { "cell_type": "markdown", - "id": "aaa71fca", + "id": "ac38d462", "metadata": { "editable": true }, @@ -922,7 +922,7 @@ }, { "cell_type": "markdown", - "id": "83af27c8", + "id": "65488a60", "metadata": { "editable": true }, @@ -935,7 +935,7 @@ }, { "cell_type": "markdown", - "id": "e22a05bb", + "id": "9c8686d8", "metadata": { "editable": true }, @@ -947,7 +947,7 @@ }, { "cell_type": "markdown", - "id": "2577f9d3", + "id": "585dbaff", "metadata": { "editable": true }, @@ -960,7 +960,7 @@ }, { "cell_type": "markdown", - "id": "244731e1", + "id": "a8093a3e", "metadata": { "editable": true }, @@ -971,7 +971,7 @@ }, { "cell_type": "markdown", - "id": "37657b97", + "id": "ec844baa", "metadata": { "editable": true }, @@ -985,7 +985,7 @@ }, { "cell_type": "markdown", - "id": "e2c5221a", + "id": "7f2b563b", "metadata": { "editable": true }, @@ -995,7 +995,7 @@ }, { "cell_type": "markdown", - "id": "68bcc41e", + "id": "5a75bbe4", "metadata": { "editable": true }, @@ -1009,7 +1009,7 @@ }, { "cell_type": "markdown", - "id": "bf00fd90", + "id": "8d00a9ff", "metadata": { "editable": true }, @@ -1022,7 +1022,7 @@ }, { "cell_type": "markdown", - "id": "27c016a4", + "id": "cae54a40", "metadata": { "editable": true }, @@ -1035,7 +1035,7 @@ }, { "cell_type": "markdown", - "id": "a9ab9ec0", + "id": "ef87a15a", "metadata": { "editable": true }, @@ -1045,7 +1045,7 @@ }, { "cell_type": "markdown", - "id": "016819e4", + "id": "5bfaef5d", "metadata": { "editable": true }, @@ -1058,7 +1058,7 @@ }, { "cell_type": "markdown", - "id": "e5d0c294", + "id": "0aab5f59", "metadata": { "editable": true }, @@ -1068,7 +1068,7 @@ }, { "cell_type": "markdown", - "id": "c07c8ef9", + "id": "71269afb", "metadata": { "editable": true }, @@ -1081,7 +1081,7 @@ }, { "cell_type": "markdown", - "id": "ca560de7", + "id": "4d809fab", "metadata": { "editable": true }, @@ -1093,7 +1093,7 @@ }, { "cell_type": "markdown", - "id": "0b3fb5f5", + "id": "734f7c5e", "metadata": { "editable": true }, @@ -1112,7 +1112,7 @@ }, { "cell_type": "markdown", - "id": "0ee945bc", + "id": "f4ff2861", "metadata": { "editable": true }, @@ -1125,7 +1125,7 @@ }, { "cell_type": "markdown", - "id": "5e0d7232", + "id": "fd66a47c", "metadata": { "editable": true }, @@ -1137,7 +1137,7 @@ }, { "cell_type": "markdown", - "id": "4f0c89e3", + "id": "51e5434b", "metadata": { "editable": true }, @@ -1150,7 +1150,7 @@ }, { "cell_type": "markdown", - "id": "c2a810e6", + "id": "9129f7a5", "metadata": { "editable": true }, @@ -1170,7 +1170,7 @@ }, { "cell_type": "markdown", - "id": "58260444", + "id": "e5fa7f9c", "metadata": { "editable": true }, @@ -1179,13 +1179,13 @@ "\n", "Confidence intervals are used in statistics and represent a type of estimate\n", "computed from the observed data. This gives a range of values for an\n", - "unknown parameter such as the parameters $\\boldsymbol{\\beta}$ from linear regression.\n", + "unknown parameter such as the parameters $\\boldsymbol{\\theta}$ from linear regression.\n", "\n", - "With the OLS expressions for the parameters $\\boldsymbol{\\beta}$ we found \n", - "$\\mathbb{E}(\\boldsymbol{\\beta}) = \\boldsymbol{\\beta}$, which means that the estimator of the regression parameters is unbiased.\n", + "With the OLS expressions for the parameters $\\boldsymbol{\\theta}$ we found \n", + "$\\mathbb{E}(\\boldsymbol{\\theta}) = \\boldsymbol{\\theta}$, which means that the estimator of the regression parameters is unbiased.\n", "\n", "In the exercises this week we show that the variance of the estimate of the $j$-th regression coefficient is\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", "\n", "This quantity can be used to\n", "construct a confidence interval for the estimates." @@ -1193,34 +1193,34 @@ }, { "cell_type": "markdown", - "id": "24ddf9fe", + "id": "f5bbafe6", "metadata": { "editable": true }, "source": [ "## Standard Approach based on the Normal Distribution\n", "\n", - "We will assume that the parameters $\\beta$ follow a normal\n", + "We will assume that the parameters $\\theta$ follow a normal\n", "distribution. We can then define the confidence interval. Here we will be using as\n", - "shorthands $\\mu_{\\beta}$ for the above mean value and $\\sigma_{\\beta}$\n", + "shorthands $\\mu_{\\theta}$ for the above mean value and $\\sigma_{\\theta}$\n", "for the standard deviation. We have then a confidence interval" ] }, { "cell_type": "markdown", - "id": "459e4649", + "id": "0cc93a0c", "metadata": { "editable": true }, "source": [ "$$\n", - "\\left(\\mu_{\\beta}\\pm \\frac{z\\sigma_{\\beta}}{\\sqrt{n}}\\right),\n", + "\\left(\\mu_{\\theta}\\pm \\frac{z\\sigma_{\\theta}}{\\sqrt{n}}\\right),\n", "$$" ] }, { "cell_type": "markdown", - "id": "8924d816", + "id": "8dd4f5e0", "metadata": { "editable": true }, @@ -1240,18 +1240,18 @@ }, { "cell_type": "markdown", - "id": "82ec4b91", + "id": "f51c546c", "metadata": { "editable": true }, "source": [ "## Resampling methods: Bootstrap background\n", "\n", - "Since $\\widehat{\\beta} = \\widehat{\\beta}(\\boldsymbol{X})$ is a function of random variables,\n", - "$\\widehat{\\beta}$ itself must be a random variable. Thus it has\n", + "Since $\\widehat{\\theta} = \\widehat{\\theta}(\\boldsymbol{X})$ is a function of random variables,\n", + "$\\widehat{\\theta}$ itself must be a random variable. Thus it has\n", "a pdf, call this function $p(\\boldsymbol{t})$. The aim of the bootstrap is to\n", "estimate $p(\\boldsymbol{t})$ by the relative frequency of\n", - "$\\widehat{\\beta}$. You can think of this as using a histogram\n", + "$\\widehat{\\theta}$. You can think of this as using a histogram\n", "in the place of $p(\\boldsymbol{t})$. If the relative frequency closely\n", "resembles $p(\\vec{t})$, then using numerics, it is straight forward to\n", "estimate all the interesting parameters of $p(\\boldsymbol{t})$ using point\n", @@ -1260,31 +1260,31 @@ }, { "cell_type": "markdown", - "id": "6e194c67", + "id": "e44fcf6d", "metadata": { "editable": true }, "source": [ "## Resampling methods: More Bootstrap background\n", "\n", - "In the case that $\\widehat{\\beta}$ has\n", + "In the case that $\\widehat{\\theta}$ has\n", "more than one component, and the components are independent, we use the\n", "same estimator on each component separately. If the probability\n", "density function of $X_i$, $p(x)$, had been known, then it would have\n", "been straightforward to do this by: \n", "1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \\cdots, X_n^*)$. \n", "\n", - "2. Then using these numbers, we could compute a replica of $\\widehat{\\beta}$ called $\\widehat{\\beta}^*$. \n", + "2. Then using these numbers, we could compute a replica of $\\widehat{\\theta}$ called $\\widehat{\\theta}^*$. \n", "\n", "By repeated use of the above two points, many\n", - "estimates of $\\widehat{\\beta}$ can be obtained. The\n", - "idea is to use the relative frequency of $\\widehat{\\beta}^*$\n", + "estimates of $\\widehat{\\theta}$ can be obtained. The\n", + "idea is to use the relative frequency of $\\widehat{\\theta}^*$\n", "(think of a histogram) as an estimate of $p(\\boldsymbol{t})$." ] }, { "cell_type": "markdown", - "id": "af32bd4f", + "id": "3bd69373", "metadata": { "editable": true }, @@ -1305,7 +1305,7 @@ }, { "cell_type": "markdown", - "id": "66865206", + "id": "e7f867d9", "metadata": { "editable": true }, @@ -1318,24 +1318,24 @@ "\n", "2. Define a vector $\\boldsymbol{x}^*$ containing the values which were drawn from $\\boldsymbol{x}$. \n", "\n", - "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\beta}^*$ by evaluating $\\widehat \\beta$ under the observations $\\boldsymbol{x}^*$. \n", + "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\theta}^*$ by evaluating $\\widehat \\theta$ under the observations $\\boldsymbol{x}^*$. \n", "\n", "4. Repeat this process $k$ times. \n", "\n", "When you are done, you can draw a histogram of the relative frequency\n", - "of $\\widehat \\beta^*$. This is your estimate of the probability\n", + "of $\\widehat \\theta^*$. This is your estimate of the probability\n", "distribution $p(t)$. Using this probability distribution you can\n", "estimate any statistics thereof. In principle you never draw the\n", - "histogram of the relative frequency of $\\widehat{\\beta}^*$. Instead\n", + "histogram of the relative frequency of $\\widehat{\\theta}^*$. Instead\n", "you use the estimators corresponding to the statistic of interest. For\n", "example, if you are interested in estimating the variance of $\\widehat\n", - "\\beta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", - "$\\widehat \\beta^*$." + "\\theta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", + "$\\widehat \\theta^*$." ] }, { "cell_type": "markdown", - "id": "c9c40149", + "id": "5c2c3909", "metadata": { "editable": true }, @@ -1359,7 +1359,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "e6ac04cd", + "id": "a32faf6a", "metadata": { "collapsed": false, "editable": true @@ -1398,7 +1398,7 @@ }, { "cell_type": "markdown", - "id": "b490ce43", + "id": "bc95505d", "metadata": { "editable": true }, @@ -1408,7 +1408,7 @@ }, { "cell_type": "markdown", - "id": "ce2afb67", + "id": "355f62af", "metadata": { "editable": true }, @@ -1419,7 +1419,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "b194e79e", + "id": "39fd1fa8", "metadata": { "collapsed": false, "editable": true @@ -1439,7 +1439,7 @@ }, { "cell_type": "markdown", - "id": "e557245e", + "id": "237b4db3", "metadata": { "editable": true }, @@ -1457,7 +1457,7 @@ }, { "cell_type": "markdown", - "id": "c30637d8", + "id": "0c13e5b0", "metadata": { "editable": true }, @@ -1469,7 +1469,7 @@ }, { "cell_type": "markdown", - "id": "e5c68caa", + "id": "a086aa7e", "metadata": { "editable": true }, @@ -1478,27 +1478,27 @@ "\n", "In our derivation of the ordinary least squares method we defined then\n", "an approximation to the function $f$ in terms of the parameters\n", - "$\\boldsymbol{\\beta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", - "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\beta}$. \n", + "$\\boldsymbol{\\theta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", + "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\theta}$. \n", "\n", - "Thereafter we found the parameters $\\boldsymbol{\\beta}$ by optimizing the means squared error via the so-called cost function" + "Thereafter we found the parameters $\\boldsymbol{\\theta}$ by optimizing the means squared error via the so-called cost function" ] }, { "cell_type": "markdown", - "id": "d0c891d5", + "id": "c3837d89", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{X},\\boldsymbol{\\beta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", + "C(\\boldsymbol{X},\\boldsymbol{\\theta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", "$$" ] }, { "cell_type": "markdown", - "id": "eb41d38c", + "id": "7c7cd0a7", "metadata": { "editable": true }, @@ -1508,7 +1508,7 @@ }, { "cell_type": "markdown", - "id": "8e74cdc8", + "id": "b9db0cd5", "metadata": { "editable": true }, @@ -1520,7 +1520,7 @@ }, { "cell_type": "markdown", - "id": "b135304d", + "id": "00482b2d", "metadata": { "editable": true }, @@ -1537,7 +1537,7 @@ }, { "cell_type": "markdown", - "id": "5fae4e71", + "id": "2e7c7291", "metadata": { "editable": true }, @@ -1549,7 +1549,7 @@ }, { "cell_type": "markdown", - "id": "4e89ca72", + "id": "9a8d29b5", "metadata": { "editable": true }, @@ -1559,7 +1559,7 @@ }, { "cell_type": "markdown", - "id": "a2edd75d", + "id": "9a8e2084", "metadata": { "editable": true }, @@ -1571,7 +1571,7 @@ }, { "cell_type": "markdown", - "id": "0dfbc890", + "id": "4a191c5c", "metadata": { "editable": true }, @@ -1581,7 +1581,7 @@ }, { "cell_type": "markdown", - "id": "4168fa86", + "id": "c37713f8", "metadata": { "editable": true }, @@ -1593,7 +1593,7 @@ }, { "cell_type": "markdown", - "id": "defe083c", + "id": "82ca4a7b", "metadata": { "editable": true }, @@ -1603,7 +1603,7 @@ }, { "cell_type": "markdown", - "id": "ed067301", + "id": "2765a841", "metadata": { "editable": true }, @@ -1619,7 +1619,7 @@ }, { "cell_type": "markdown", - "id": "815dc960", + "id": "321964c1", "metadata": { "editable": true }, @@ -1630,7 +1630,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "1ca319f9", + "id": "5942226f", "metadata": { "collapsed": false, "editable": true @@ -1695,7 +1695,7 @@ }, { "cell_type": "markdown", - "id": "7c9d4971", + "id": "cf2af19d", "metadata": { "editable": true }, @@ -1706,7 +1706,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "3b1f1345", + "id": "7b371a0c", "metadata": { "collapsed": false, "editable": true @@ -1763,7 +1763,7 @@ }, { "cell_type": "markdown", - "id": "a1dfc4a3", + "id": "6f583fcd", "metadata": { "editable": true }, @@ -1801,7 +1801,7 @@ }, { "cell_type": "markdown", - "id": "1b0835bb", + "id": "37766c51", "metadata": { "editable": true }, @@ -1828,7 +1828,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "3c6f8392", + "id": "0eac5ca9", "metadata": { "collapsed": false, "editable": true @@ -1890,7 +1890,7 @@ }, { "cell_type": "markdown", - "id": "195d6e77", + "id": "d698096e", "metadata": { "editable": true }, @@ -1915,7 +1915,7 @@ }, { "cell_type": "markdown", - "id": "1458a723", + "id": "3e37a4d2", "metadata": { "editable": true }, @@ -1943,7 +1943,7 @@ }, { "cell_type": "markdown", - "id": "d5a79ca2", + "id": "e7e43cf9", "metadata": { "editable": true }, @@ -1956,7 +1956,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "65b5aaec", + "id": "7aa6a568", "metadata": { "collapsed": false, "editable": true @@ -2056,7 +2056,7 @@ }, { "cell_type": "markdown", - "id": "d57db05f", + "id": "9a90aec7", "metadata": { "editable": true }, @@ -2067,7 +2067,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "bf4fcde2", + "id": "93b3af24", "metadata": { "collapsed": false, "editable": true @@ -2156,7 +2156,7 @@ }, { "cell_type": "markdown", - "id": "ea3509c6", + "id": "90657c6d", "metadata": { "editable": true }, @@ -2166,7 +2166,7 @@ }, { "cell_type": "markdown", - "id": "95edcb8b", + "id": "a88b86e7", "metadata": { "editable": true }, @@ -2179,7 +2179,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "efed8b4e", + "id": "89f45188", "metadata": { "collapsed": false, "editable": true @@ -2257,7 +2257,7 @@ }, { "cell_type": "markdown", - "id": "8dadd613", + "id": "75496517", "metadata": { "editable": true }, diff --git a/doc/LectureNotes/week38.ipynb b/doc/LectureNotes/week38.ipynb index c474d1c83..544d286d0 100644 --- a/doc/LectureNotes/week38.ipynb +++ b/doc/LectureNotes/week38.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "3896fc4b", + "id": "169923b3", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "ec008b77", + "id": "47013ee7", "metadata": { "editable": true }, @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "997c17fc", + "id": "d1fb5464", "metadata": { "editable": true }, @@ -47,7 +47,7 @@ }, { "cell_type": "markdown", - "id": "893cd04d", + "id": "1a34a7ce", "metadata": { "editable": true }, @@ -68,7 +68,7 @@ }, { "cell_type": "markdown", - "id": "115e99b5", + "id": "c6f56f83", "metadata": { "editable": true }, @@ -81,7 +81,7 @@ "particular, we will focus on what the regularization terms can result\n", "in. We will amongst other things show that the regularization\n", "parameter can reduce considerably the variance of the parameters\n", - "$\\beta$.\n", + "$\\theta$.\n", "\n", "On of the advantages of doing linear regression is that we actually end up with\n", "analytical expressions for several statistical quantities. \n", @@ -96,7 +96,7 @@ }, { "cell_type": "markdown", - "id": "857956f2", + "id": "3ece0c04", "metadata": { "editable": true }, @@ -112,7 +112,7 @@ }, { "cell_type": "markdown", - "id": "074e7f9c", + "id": "d36cf6db", "metadata": { "editable": true }, @@ -120,7 +120,7 @@ "The randomness of $\\varepsilon_i$ implies that\n", "$\\mathbf{y}_i$ is also a random variable. In particular,\n", "$\\mathbf{y}_i$ is normally distributed, because $\\varepsilon_i \\sim\n", - "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\beta}$ is a\n", + "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\theta}$ is a\n", "non-random scalar. To specify the parameters of the distribution of\n", "$\\mathbf{y}_i$ we need to calculate its first two moments. \n", "\n", @@ -131,7 +131,7 @@ }, { "cell_type": "markdown", - "id": "0220f12c", + "id": "5903d2be", "metadata": { "editable": true }, @@ -145,7 +145,7 @@ }, { "cell_type": "markdown", - "id": "fe96cf24", + "id": "37f4e199", "metadata": { "editable": true }, @@ -157,7 +157,7 @@ }, { "cell_type": "markdown", - "id": "b308e3f9", + "id": "7a7b084f", "metadata": { "editable": true }, @@ -168,19 +168,19 @@ }, { "cell_type": "markdown", - "id": "a2d28d54", + "id": "1ec511a1", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta}.\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "501243ea", + "id": "ffe76935", "metadata": { "editable": true }, @@ -192,7 +192,7 @@ }, { "cell_type": "markdown", - "id": "a0735fd3", + "id": "e569274a", "metadata": { "editable": true }, @@ -200,15 +200,15 @@ "$$\n", "\\begin{align*} \n", "\\mathbb{E}(y_i) & =\n", - "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}) + \\mathbb{E}(\\varepsilon_i)\n", - "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\beta, \n", + "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}) + \\mathbb{E}(\\varepsilon_i)\n", + "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\theta, \n", "\\end{align*}\n", "$$" ] }, { "cell_type": "markdown", - "id": "105564c8", + "id": "c4dd2623", "metadata": { "editable": true }, @@ -219,7 +219,7 @@ }, { "cell_type": "markdown", - "id": "e484179a", + "id": "df2f8936", "metadata": { "editable": true }, @@ -228,12 +228,12 @@ "\\begin{align*} \\mbox{Var}(y_i) & = \\mathbb{E} \\{ [y_i\n", "- \\mathbb{E}(y_i)]^2 \\} \\, \\, \\, = \\, \\, \\, \\mathbb{E} ( y_i^2 ) -\n", "[\\mathbb{E}(y_i)]^2 \\\\ & = \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\,\n", - "\\beta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \\\\ &\n", - "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2 \\varepsilon_i\n", - "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", - "\\ast} \\, \\beta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2\n", - "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} +\n", - "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \n", + "\\theta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \\\\ &\n", + "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2 \\varepsilon_i\n", + "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", + "\\ast} \\, \\theta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2\n", + "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} +\n", + "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \n", "\\\\ & = \\mathbb{E}(\\varepsilon_i^2 ) \\, \\, \\, = \\, \\, \\,\n", "\\mbox{Var}(\\varepsilon_i) \\, \\, \\, = \\, \\, \\, \\sigma^2. \n", "\\end{align*}\n", @@ -242,42 +242,42 @@ }, { "cell_type": "markdown", - "id": "028be304", + "id": "0a3e5956", "metadata": { "editable": true }, "source": [ - "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", - "mean value $\\boldsymbol{X}\\boldsymbol{\\beta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." + "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", + "mean value $\\boldsymbol{X}\\boldsymbol{\\theta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." ] }, { "cell_type": "markdown", - "id": "71256ba5", + "id": "973e45a3", "metadata": { "editable": true }, "source": [ - "## Expectation value and variance for $\\boldsymbol{\\beta}$\n", + "## Expectation value and variance for $\\boldsymbol{\\theta}$\n", "\n", - "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\beta}}$ we can evaluate the expectation value" + "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\theta}}$ we can evaluate the expectation value" ] }, { "cell_type": "markdown", - "id": "236fd8a6", + "id": "ff486d1e", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E}(\\boldsymbol{\\hat{\\beta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\beta}=\\boldsymbol{\\beta}.\n", + "\\mathbb{E}(\\boldsymbol{\\hat{\\theta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\theta}=\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "8d2e261d", + "id": "e4307815", "metadata": { "editable": true }, @@ -286,35 +286,35 @@ "\n", "We can also calculate the variance\n", "\n", - "The variance of the optimal value $\\boldsymbol{\\hat{\\beta}}$ is" + "The variance of the optimal value $\\boldsymbol{\\hat{\\theta}}$ is" ] }, { "cell_type": "markdown", - "id": "2d8ebdef", + "id": "490b2cbf", "metadata": { "editable": true }, "source": [ "$$\n", "\\begin{eqnarray*}\n", - "\\mbox{Var}(\\boldsymbol{\\hat{\\beta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})] [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})]^{T} \\}\n", + "\\mbox{Var}(\\boldsymbol{\\hat{\\theta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})] [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})]^{T} \\}\n", "\\\\\n", - "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}]^{T} \\}\n", + "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}]^{T} \\}\n", "\\\\\n", - "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", + "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", "% \\\\\n", - "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\boldsymbol{\\beta}^T\n", + "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\boldsymbol{\\theta}^T\n", "\\\\\n", - "& = & \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\, \\, \\, = \\, \\, \\, \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1},\n", "\\end{eqnarray*}\n", "$$" @@ -322,21 +322,21 @@ }, { "cell_type": "markdown", - "id": "3285f57f", + "id": "5d8dd6bc", "metadata": { "editable": true }, "source": [ "where we have used that $\\mathbb{E} (\\mathbf{Y} \\mathbf{Y}^{T}) =\n", - "\\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} +\n", - "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\beta}) = \\sigma^2\n", + "\\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} +\n", + "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\theta}) = \\sigma^2\n", "\\, (\\mathbf{X}^{T} \\mathbf{X})^{-1}$, one obtains an estimate of the\n", "variance of the estimate of the $j$-th regression coefficient:\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", "construct a confidence interval for the estimates.\n", "\n", "In a similar way, we can obtain analytical expressions for say the\n", - "expectation values of the parameters $\\boldsymbol{\\beta}$ and their variance\n", + "expectation values of the parameters $\\boldsymbol{\\theta}$ and their variance\n", "when we employ Ridge regression, allowing us again to define a confidence interval. \n", "\n", "It is rather straightforward to show that" @@ -344,80 +344,80 @@ }, { "cell_type": "markdown", - "id": "925312c4", + "id": "3594078a", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\beta}^{\\mathrm{OLS}}.\n", + "\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\theta}^{\\mathrm{OLS}}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "b9fd6bed", + "id": "43007b7a", "metadata": { "editable": true }, "source": [ "We see clearly that \n", - "$\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\beta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", + "$\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\theta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", "\n", "We can also compute the variance as" ] }, { "cell_type": "markdown", - "id": "d919b998", + "id": "3cd4e7da", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", "$$" ] }, { "cell_type": "markdown", - "id": "a878a50d", + "id": "5816b21e", "metadata": { "editable": true }, "source": [ - "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\beta}$ goes to zero. \n", + "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\theta}$ goes to zero. \n", "\n", "With this, we can compute the difference" ] }, { "cell_type": "markdown", - "id": "ccbbea14", + "id": "b58b8607", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\beta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\theta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "74775fd2", + "id": "c843db7f", "metadata": { "editable": true }, "source": [ "The difference is non-negative definite since each component of the\n", "matrix product is non-negative definite. \n", - "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." + "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\theta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." ] }, { "cell_type": "markdown", - "id": "032b9317", + "id": "e81b1c39", "metadata": { "editable": true }, @@ -431,28 +431,28 @@ "$\\sigma^2$.\n", "\n", "We found above that the outputs $\\boldsymbol{y}$ have a mean value given by\n", - "$\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$ and variance $\\sigma^2$. Since the entries to\n", + "$\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$ and variance $\\sigma^2$. Since the entries to\n", "the design matrix are not stochastic variables, we can assume that the\n", "probability distribution of our targets is also a normal distribution\n", - "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$. This means that a\n", + "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$. This means that a\n", "single output $y_i$ is given by the Gaussian distribution" ] }, { "cell_type": "markdown", - "id": "1c61251d", + "id": "ea9719e3", "metadata": { "editable": true }, "source": [ "$$\n", - "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "5545f5c8", + "id": "c28b598f", "metadata": { "editable": true }, @@ -465,43 +465,43 @@ }, { "cell_type": "markdown", - "id": "470b75b2", + "id": "88063ed7", "metadata": { "editable": true }, "source": [ "$$\n", - "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]},\n", + "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]},\n", "$$" ] }, { "cell_type": "markdown", - "id": "908e209b", + "id": "a34e4c28", "metadata": { "editable": true }, "source": [ - "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\beta}$.\n", + "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\theta}$.\n", "\n", "Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event $\\boldsymbol{y}$ as the product of the single events, that is we have" ] }, { "cell_type": "markdown", - "id": "4fdbf729", + "id": "1fc7ae0a", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta}).\n", + "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta}).\n", "$$" ] }, { "cell_type": "markdown", - "id": "ac5dc52a", + "id": "30d5a1be", "metadata": { "editable": true }, @@ -512,7 +512,7 @@ }, { "cell_type": "markdown", - "id": "dc35e3e1", + "id": "1d29fd29", "metadata": { "editable": true }, @@ -524,7 +524,7 @@ }, { "cell_type": "markdown", - "id": "f0b84ac4", + "id": "289a116c", "metadata": { "editable": true }, @@ -535,29 +535,29 @@ }, { "cell_type": "markdown", - "id": "90eb85ae", + "id": "d1cfbb56", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{D}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "p(\\boldsymbol{D}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "5e70b7b1", + "id": "e089fc0e", "metadata": { "editable": true }, "source": [ - "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\beta}$." + "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\theta}$." ] }, { "cell_type": "markdown", - "id": "dcf0e487", + "id": "32cf9944", "metadata": { "editable": true }, @@ -571,7 +571,7 @@ "data is the most probable. \n", "\n", "We will assume here that our events are given by the above Gaussian\n", - "distribution and we will determine the optimal parameters $\\beta$ by\n", + "distribution and we will determine the optimal parameters $\\theta$ by\n", "maximizing the above PDF. However, computing the derivatives of a\n", "product function is cumbersome and can easily lead to overflow and/or\n", "underflowproblems, with potentials for loss of numerical precision.\n", @@ -588,7 +588,7 @@ }, { "cell_type": "markdown", - "id": "c14bbd71", + "id": "ee8a9544", "metadata": { "editable": true }, @@ -600,19 +600,19 @@ }, { "cell_type": "markdown", - "id": "13fcb784", + "id": "4bfeb203", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})},\n", + "C(\\boldsymbol{\\theta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})},\n", "$$" ] }, { "cell_type": "markdown", - "id": "8f685501", + "id": "256143f9", "metadata": { "editable": true }, @@ -622,63 +622,63 @@ }, { "cell_type": "markdown", - "id": "1b83ae5f", + "id": "60d75bb1", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})\\vert\\vert_2^2}{2\\sigma^2}.\n", + "C(\\boldsymbol{\\theta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta})\\vert\\vert_2^2}{2\\sigma^2}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "c44b7952", + "id": "7b111014", "metadata": { "editable": true }, "source": [ - "Taking the derivative of the *new* cost function with respect to the parameters $\\beta$ we recognize our familiar OLS equation, namely" + "Taking the derivative of the *new* cost function with respect to the parameters $\\theta$ we recognize our familiar OLS equation, namely" ] }, { "cell_type": "markdown", - "id": "7848c18a", + "id": "0f42a0b5", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right) =0,\n", + "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta}\\right) =0,\n", "$$" ] }, { "cell_type": "markdown", - "id": "bd67b48f", + "id": "39a9276b", "metadata": { "editable": true }, "source": [ - "which leads to the well-known OLS equation for the optimal paramters $\\beta$" + "which leads to the well-known OLS equation for the optimal paramters $\\theta$" ] }, { "cell_type": "markdown", - "id": "8a8665d9", + "id": "84c63927", "metadata": { "editable": true }, "source": [ "$$\n", - "\\hat{\\boldsymbol{\\beta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", + "\\hat{\\boldsymbol{\\theta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", "$$" ] }, { "cell_type": "markdown", - "id": "718fe69c", + "id": "c087ef22", "metadata": { "editable": true }, @@ -688,7 +688,7 @@ }, { "cell_type": "markdown", - "id": "f837afb3", + "id": "79503987", "metadata": { "editable": true }, @@ -707,7 +707,7 @@ }, { "cell_type": "markdown", - "id": "82ed1dd7", + "id": "c165025b", "metadata": { "editable": true }, @@ -735,7 +735,7 @@ }, { "cell_type": "markdown", - "id": "06046f8f", + "id": "efb63405", "metadata": { "editable": true }, @@ -761,7 +761,7 @@ }, { "cell_type": "markdown", - "id": "7639ed0d", + "id": "88d4bfd8", "metadata": { "editable": true }, @@ -778,7 +778,7 @@ }, { "cell_type": "markdown", - "id": "dc080604", + "id": "6a0795e0", "metadata": { "editable": true }, @@ -798,7 +798,7 @@ }, { "cell_type": "markdown", - "id": "11aa170a", + "id": "d815fdf3", "metadata": { "editable": true }, @@ -827,7 +827,7 @@ }, { "cell_type": "markdown", - "id": "a56444a1", + "id": "a5b26ed1", "metadata": { "editable": true }, @@ -852,7 +852,7 @@ }, { "cell_type": "markdown", - "id": "020c8c0c", + "id": "e5783b81", "metadata": { "editable": true }, @@ -872,7 +872,7 @@ }, { "cell_type": "markdown", - "id": "f707010c", + "id": "9bdfff4f", "metadata": { "editable": true }, @@ -884,7 +884,7 @@ }, { "cell_type": "markdown", - "id": "062d8d7c", + "id": "bd1e4a83", "metadata": { "editable": true }, @@ -894,7 +894,7 @@ }, { "cell_type": "markdown", - "id": "adf6bd14", + "id": "66d14a83", "metadata": { "editable": true }, @@ -909,7 +909,7 @@ }, { "cell_type": "markdown", - "id": "aaa71fca", + "id": "ac38d462", "metadata": { "editable": true }, @@ -922,7 +922,7 @@ }, { "cell_type": "markdown", - "id": "83af27c8", + "id": "65488a60", "metadata": { "editable": true }, @@ -935,7 +935,7 @@ }, { "cell_type": "markdown", - "id": "e22a05bb", + "id": "9c8686d8", "metadata": { "editable": true }, @@ -947,7 +947,7 @@ }, { "cell_type": "markdown", - "id": "2577f9d3", + "id": "585dbaff", "metadata": { "editable": true }, @@ -960,7 +960,7 @@ }, { "cell_type": "markdown", - "id": "244731e1", + "id": "a8093a3e", "metadata": { "editable": true }, @@ -971,7 +971,7 @@ }, { "cell_type": "markdown", - "id": "37657b97", + "id": "ec844baa", "metadata": { "editable": true }, @@ -985,7 +985,7 @@ }, { "cell_type": "markdown", - "id": "e2c5221a", + "id": "7f2b563b", "metadata": { "editable": true }, @@ -995,7 +995,7 @@ }, { "cell_type": "markdown", - "id": "68bcc41e", + "id": "5a75bbe4", "metadata": { "editable": true }, @@ -1009,7 +1009,7 @@ }, { "cell_type": "markdown", - "id": "bf00fd90", + "id": "8d00a9ff", "metadata": { "editable": true }, @@ -1022,7 +1022,7 @@ }, { "cell_type": "markdown", - "id": "27c016a4", + "id": "cae54a40", "metadata": { "editable": true }, @@ -1035,7 +1035,7 @@ }, { "cell_type": "markdown", - "id": "a9ab9ec0", + "id": "ef87a15a", "metadata": { "editable": true }, @@ -1045,7 +1045,7 @@ }, { "cell_type": "markdown", - "id": "016819e4", + "id": "5bfaef5d", "metadata": { "editable": true }, @@ -1058,7 +1058,7 @@ }, { "cell_type": "markdown", - "id": "e5d0c294", + "id": "0aab5f59", "metadata": { "editable": true }, @@ -1068,7 +1068,7 @@ }, { "cell_type": "markdown", - "id": "c07c8ef9", + "id": "71269afb", "metadata": { "editable": true }, @@ -1081,7 +1081,7 @@ }, { "cell_type": "markdown", - "id": "ca560de7", + "id": "4d809fab", "metadata": { "editable": true }, @@ -1093,7 +1093,7 @@ }, { "cell_type": "markdown", - "id": "0b3fb5f5", + "id": "734f7c5e", "metadata": { "editable": true }, @@ -1112,7 +1112,7 @@ }, { "cell_type": "markdown", - "id": "0ee945bc", + "id": "f4ff2861", "metadata": { "editable": true }, @@ -1125,7 +1125,7 @@ }, { "cell_type": "markdown", - "id": "5e0d7232", + "id": "fd66a47c", "metadata": { "editable": true }, @@ -1137,7 +1137,7 @@ }, { "cell_type": "markdown", - "id": "4f0c89e3", + "id": "51e5434b", "metadata": { "editable": true }, @@ -1150,7 +1150,7 @@ }, { "cell_type": "markdown", - "id": "c2a810e6", + "id": "9129f7a5", "metadata": { "editable": true }, @@ -1170,7 +1170,7 @@ }, { "cell_type": "markdown", - "id": "58260444", + "id": "e5fa7f9c", "metadata": { "editable": true }, @@ -1179,13 +1179,13 @@ "\n", "Confidence intervals are used in statistics and represent a type of estimate\n", "computed from the observed data. This gives a range of values for an\n", - "unknown parameter such as the parameters $\\boldsymbol{\\beta}$ from linear regression.\n", + "unknown parameter such as the parameters $\\boldsymbol{\\theta}$ from linear regression.\n", "\n", - "With the OLS expressions for the parameters $\\boldsymbol{\\beta}$ we found \n", - "$\\mathbb{E}(\\boldsymbol{\\beta}) = \\boldsymbol{\\beta}$, which means that the estimator of the regression parameters is unbiased.\n", + "With the OLS expressions for the parameters $\\boldsymbol{\\theta}$ we found \n", + "$\\mathbb{E}(\\boldsymbol{\\theta}) = \\boldsymbol{\\theta}$, which means that the estimator of the regression parameters is unbiased.\n", "\n", "In the exercises this week we show that the variance of the estimate of the $j$-th regression coefficient is\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", "\n", "This quantity can be used to\n", "construct a confidence interval for the estimates." @@ -1193,34 +1193,34 @@ }, { "cell_type": "markdown", - "id": "24ddf9fe", + "id": "f5bbafe6", "metadata": { "editable": true }, "source": [ "## Standard Approach based on the Normal Distribution\n", "\n", - "We will assume that the parameters $\\beta$ follow a normal\n", + "We will assume that the parameters $\\theta$ follow a normal\n", "distribution. We can then define the confidence interval. Here we will be using as\n", - "shorthands $\\mu_{\\beta}$ for the above mean value and $\\sigma_{\\beta}$\n", + "shorthands $\\mu_{\\theta}$ for the above mean value and $\\sigma_{\\theta}$\n", "for the standard deviation. We have then a confidence interval" ] }, { "cell_type": "markdown", - "id": "459e4649", + "id": "0cc93a0c", "metadata": { "editable": true }, "source": [ "$$\n", - "\\left(\\mu_{\\beta}\\pm \\frac{z\\sigma_{\\beta}}{\\sqrt{n}}\\right),\n", + "\\left(\\mu_{\\theta}\\pm \\frac{z\\sigma_{\\theta}}{\\sqrt{n}}\\right),\n", "$$" ] }, { "cell_type": "markdown", - "id": "8924d816", + "id": "8dd4f5e0", "metadata": { "editable": true }, @@ -1240,18 +1240,18 @@ }, { "cell_type": "markdown", - "id": "82ec4b91", + "id": "f51c546c", "metadata": { "editable": true }, "source": [ "## Resampling methods: Bootstrap background\n", "\n", - "Since $\\widehat{\\beta} = \\widehat{\\beta}(\\boldsymbol{X})$ is a function of random variables,\n", - "$\\widehat{\\beta}$ itself must be a random variable. Thus it has\n", + "Since $\\widehat{\\theta} = \\widehat{\\theta}(\\boldsymbol{X})$ is a function of random variables,\n", + "$\\widehat{\\theta}$ itself must be a random variable. Thus it has\n", "a pdf, call this function $p(\\boldsymbol{t})$. The aim of the bootstrap is to\n", "estimate $p(\\boldsymbol{t})$ by the relative frequency of\n", - "$\\widehat{\\beta}$. You can think of this as using a histogram\n", + "$\\widehat{\\theta}$. You can think of this as using a histogram\n", "in the place of $p(\\boldsymbol{t})$. If the relative frequency closely\n", "resembles $p(\\vec{t})$, then using numerics, it is straight forward to\n", "estimate all the interesting parameters of $p(\\boldsymbol{t})$ using point\n", @@ -1260,31 +1260,31 @@ }, { "cell_type": "markdown", - "id": "6e194c67", + "id": "e44fcf6d", "metadata": { "editable": true }, "source": [ "## Resampling methods: More Bootstrap background\n", "\n", - "In the case that $\\widehat{\\beta}$ has\n", + "In the case that $\\widehat{\\theta}$ has\n", "more than one component, and the components are independent, we use the\n", "same estimator on each component separately. If the probability\n", "density function of $X_i$, $p(x)$, had been known, then it would have\n", "been straightforward to do this by: \n", "1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \\cdots, X_n^*)$. \n", "\n", - "2. Then using these numbers, we could compute a replica of $\\widehat{\\beta}$ called $\\widehat{\\beta}^*$. \n", + "2. Then using these numbers, we could compute a replica of $\\widehat{\\theta}$ called $\\widehat{\\theta}^*$. \n", "\n", "By repeated use of the above two points, many\n", - "estimates of $\\widehat{\\beta}$ can be obtained. The\n", - "idea is to use the relative frequency of $\\widehat{\\beta}^*$\n", + "estimates of $\\widehat{\\theta}$ can be obtained. The\n", + "idea is to use the relative frequency of $\\widehat{\\theta}^*$\n", "(think of a histogram) as an estimate of $p(\\boldsymbol{t})$." ] }, { "cell_type": "markdown", - "id": "af32bd4f", + "id": "3bd69373", "metadata": { "editable": true }, @@ -1305,7 +1305,7 @@ }, { "cell_type": "markdown", - "id": "66865206", + "id": "e7f867d9", "metadata": { "editable": true }, @@ -1318,24 +1318,24 @@ "\n", "2. Define a vector $\\boldsymbol{x}^*$ containing the values which were drawn from $\\boldsymbol{x}$. \n", "\n", - "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\beta}^*$ by evaluating $\\widehat \\beta$ under the observations $\\boldsymbol{x}^*$. \n", + "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\theta}^*$ by evaluating $\\widehat \\theta$ under the observations $\\boldsymbol{x}^*$. \n", "\n", "4. Repeat this process $k$ times. \n", "\n", "When you are done, you can draw a histogram of the relative frequency\n", - "of $\\widehat \\beta^*$. This is your estimate of the probability\n", + "of $\\widehat \\theta^*$. This is your estimate of the probability\n", "distribution $p(t)$. Using this probability distribution you can\n", "estimate any statistics thereof. In principle you never draw the\n", - "histogram of the relative frequency of $\\widehat{\\beta}^*$. Instead\n", + "histogram of the relative frequency of $\\widehat{\\theta}^*$. Instead\n", "you use the estimators corresponding to the statistic of interest. For\n", "example, if you are interested in estimating the variance of $\\widehat\n", - "\\beta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", - "$\\widehat \\beta^*$." + "\\theta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", + "$\\widehat \\theta^*$." ] }, { "cell_type": "markdown", - "id": "c9c40149", + "id": "5c2c3909", "metadata": { "editable": true }, @@ -1359,7 +1359,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "e6ac04cd", + "id": "a32faf6a", "metadata": { "collapsed": false, "editable": true @@ -1398,7 +1398,7 @@ }, { "cell_type": "markdown", - "id": "b490ce43", + "id": "bc95505d", "metadata": { "editable": true }, @@ -1408,7 +1408,7 @@ }, { "cell_type": "markdown", - "id": "ce2afb67", + "id": "355f62af", "metadata": { "editable": true }, @@ -1419,7 +1419,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "b194e79e", + "id": "39fd1fa8", "metadata": { "collapsed": false, "editable": true @@ -1439,7 +1439,7 @@ }, { "cell_type": "markdown", - "id": "e557245e", + "id": "237b4db3", "metadata": { "editable": true }, @@ -1457,7 +1457,7 @@ }, { "cell_type": "markdown", - "id": "c30637d8", + "id": "0c13e5b0", "metadata": { "editable": true }, @@ -1469,7 +1469,7 @@ }, { "cell_type": "markdown", - "id": "e5c68caa", + "id": "a086aa7e", "metadata": { "editable": true }, @@ -1478,27 +1478,27 @@ "\n", "In our derivation of the ordinary least squares method we defined then\n", "an approximation to the function $f$ in terms of the parameters\n", - "$\\boldsymbol{\\beta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", - "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\beta}$. \n", + "$\\boldsymbol{\\theta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", + "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\theta}$. \n", "\n", - "Thereafter we found the parameters $\\boldsymbol{\\beta}$ by optimizing the means squared error via the so-called cost function" + "Thereafter we found the parameters $\\boldsymbol{\\theta}$ by optimizing the means squared error via the so-called cost function" ] }, { "cell_type": "markdown", - "id": "d0c891d5", + "id": "c3837d89", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{X},\\boldsymbol{\\beta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", + "C(\\boldsymbol{X},\\boldsymbol{\\theta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", "$$" ] }, { "cell_type": "markdown", - "id": "eb41d38c", + "id": "7c7cd0a7", "metadata": { "editable": true }, @@ -1508,7 +1508,7 @@ }, { "cell_type": "markdown", - "id": "8e74cdc8", + "id": "b9db0cd5", "metadata": { "editable": true }, @@ -1520,7 +1520,7 @@ }, { "cell_type": "markdown", - "id": "b135304d", + "id": "00482b2d", "metadata": { "editable": true }, @@ -1537,7 +1537,7 @@ }, { "cell_type": "markdown", - "id": "5fae4e71", + "id": "2e7c7291", "metadata": { "editable": true }, @@ -1549,7 +1549,7 @@ }, { "cell_type": "markdown", - "id": "4e89ca72", + "id": "9a8d29b5", "metadata": { "editable": true }, @@ -1559,7 +1559,7 @@ }, { "cell_type": "markdown", - "id": "a2edd75d", + "id": "9a8e2084", "metadata": { "editable": true }, @@ -1571,7 +1571,7 @@ }, { "cell_type": "markdown", - "id": "0dfbc890", + "id": "4a191c5c", "metadata": { "editable": true }, @@ -1581,7 +1581,7 @@ }, { "cell_type": "markdown", - "id": "4168fa86", + "id": "c37713f8", "metadata": { "editable": true }, @@ -1593,7 +1593,7 @@ }, { "cell_type": "markdown", - "id": "defe083c", + "id": "82ca4a7b", "metadata": { "editable": true }, @@ -1603,7 +1603,7 @@ }, { "cell_type": "markdown", - "id": "ed067301", + "id": "2765a841", "metadata": { "editable": true }, @@ -1619,7 +1619,7 @@ }, { "cell_type": "markdown", - "id": "815dc960", + "id": "321964c1", "metadata": { "editable": true }, @@ -1630,7 +1630,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "1ca319f9", + "id": "5942226f", "metadata": { "collapsed": false, "editable": true @@ -1695,7 +1695,7 @@ }, { "cell_type": "markdown", - "id": "7c9d4971", + "id": "cf2af19d", "metadata": { "editable": true }, @@ -1706,7 +1706,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "3b1f1345", + "id": "7b371a0c", "metadata": { "collapsed": false, "editable": true @@ -1763,7 +1763,7 @@ }, { "cell_type": "markdown", - "id": "a1dfc4a3", + "id": "6f583fcd", "metadata": { "editable": true }, @@ -1801,7 +1801,7 @@ }, { "cell_type": "markdown", - "id": "1b0835bb", + "id": "37766c51", "metadata": { "editable": true }, @@ -1828,7 +1828,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "3c6f8392", + "id": "0eac5ca9", "metadata": { "collapsed": false, "editable": true @@ -1890,7 +1890,7 @@ }, { "cell_type": "markdown", - "id": "195d6e77", + "id": "d698096e", "metadata": { "editable": true }, @@ -1915,7 +1915,7 @@ }, { "cell_type": "markdown", - "id": "1458a723", + "id": "3e37a4d2", "metadata": { "editable": true }, @@ -1943,7 +1943,7 @@ }, { "cell_type": "markdown", - "id": "d5a79ca2", + "id": "e7e43cf9", "metadata": { "editable": true }, @@ -1956,7 +1956,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "65b5aaec", + "id": "7aa6a568", "metadata": { "collapsed": false, "editable": true @@ -2056,7 +2056,7 @@ }, { "cell_type": "markdown", - "id": "d57db05f", + "id": "9a90aec7", "metadata": { "editable": true }, @@ -2067,7 +2067,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "bf4fcde2", + "id": "93b3af24", "metadata": { "collapsed": false, "editable": true @@ -2156,7 +2156,7 @@ }, { "cell_type": "markdown", - "id": "ea3509c6", + "id": "90657c6d", "metadata": { "editable": true }, @@ -2166,7 +2166,7 @@ }, { "cell_type": "markdown", - "id": "95edcb8b", + "id": "a88b86e7", "metadata": { "editable": true }, @@ -2179,7 +2179,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "efed8b4e", + "id": "89f45188", "metadata": { "collapsed": false, "editable": true @@ -2257,7 +2257,7 @@ }, { "cell_type": "markdown", - "id": "8dadd613", + "id": "75496517", "metadata": { "editable": true }, diff --git a/doc/pub/week38/html/._week38-bs000.html b/doc/pub/week38/html/._week38-bs000.html index b468e462d..0c055159c 100644 --- a/doc/pub/week38/html/._week38-bs000.html +++ b/doc/pub/week38/html/._week38-bs000.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs001.html b/doc/pub/week38/html/._week38-bs001.html index 2c83127be..ed985caeb 100644 --- a/doc/pub/week38/html/._week38-bs001.html +++ b/doc/pub/week38/html/._week38-bs001.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs002.html b/doc/pub/week38/html/._week38-bs002.html index 335ebeafd..7e2918a9e 100644 --- a/doc/pub/week38/html/._week38-bs002.html +++ b/doc/pub/week38/html/._week38-bs002.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs003.html b/doc/pub/week38/html/._week38-bs003.html index 729f1adce..01a605f1d 100644 --- a/doc/pub/week38/html/._week38-bs003.html +++ b/doc/pub/week38/html/._week38-bs003.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -259,7 +259,7 @@ move from a linear algebra analysis to a statistical analysis. In particular, we will focus on what the regularization terms can result in. We will amongst other things show that the regularization parameter can reduce considerably the variance of the parameters -\( \beta \). +\( \theta \).

    On of the advantages of doing linear regression is that we actually end up with @@ -284,7 +284,7 @@ $$

    The randomness of \( \varepsilon_i \) implies that \( \mathbf{y}_i \) is also a random variable. In particular, \( \mathbf{y}_i \) is normally distributed, because \( \varepsilon_i \sim -\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\beta} \) is a +\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\theta} \) is a non-random scalar. To specify the parameters of the distribution of \( \mathbf{y}_i \) we need to calculate its first two moments.

    diff --git a/doc/pub/week38/html/._week38-bs004.html b/doc/pub/week38/html/._week38-bs004.html index 68febe9e3..33ef42b90 100644 --- a/doc/pub/week38/html/._week38-bs004.html +++ b/doc/pub/week38/html/._week38-bs004.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -265,7 +265,7 @@ $$ function \( f \) is approximated by \( \boldsymbol{\tilde{y}} \) where we want to minimize \( (\boldsymbol{y}-\boldsymbol{\tilde{y}})^2 \), our MSE, with

    $$ -\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\theta}. $$ diff --git a/doc/pub/week38/html/._week38-bs005.html b/doc/pub/week38/html/._week38-bs005.html index dc4b0349e..af0731ea8 100644 --- a/doc/pub/week38/html/._week38-bs005.html +++ b/doc/pub/week38/html/._week38-bs005.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -257,8 +257,8 @@ MathJax.Hub.Config({ $$ \begin{align*} \mathbb{E}(y_i) & = -\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) -\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\theta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \theta, \end{align*} $$ @@ -269,19 +269,19 @@ $$ \begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i - \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - [\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, -\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & -= \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i -\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, -\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 -\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + -\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\theta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \theta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, \mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. \end{align*} $$ -

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with -mean value \( \boldsymbol{X}\boldsymbol{\beta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD). +

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with +mean value \( \boldsymbol{X}\boldsymbol{\theta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD).

    diff --git a/doc/pub/week38/html/._week38-bs006.html b/doc/pub/week38/html/._week38-bs006.html index 7da46d9c2..a6dac2bd5 100644 --- a/doc/pub/week38/html/._week38-bs006.html +++ b/doc/pub/week38/html/._week38-bs006.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -251,81 +251,81 @@ MathJax.Hub.Config({

     

     

     

    -

    Expectation value and variance for \( \boldsymbol{\beta} \)

    +

    Expectation value and variance for \( \boldsymbol{\theta} \)

    -

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\beta}} \) we can evaluate the expectation value

    +

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\theta}} \) we can evaluate the expectation value

    $$ -\mathbb{E}(\boldsymbol{\hat{\beta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +\mathbb{E}(\boldsymbol{\hat{\theta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\theta}=\boldsymbol{\theta}. $$

    This means that the estimator of the regression parameters is unbiased.

    We can also calculate the variance

    -

    The variance of the optimal value \( \boldsymbol{\hat{\beta}} \) is

    +

    The variance of the optimal value \( \boldsymbol{\hat{\theta}} \) is

    $$ \begin{eqnarray*} -\mbox{Var}(\boldsymbol{\hat{\beta}}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\mbox{Var}(\boldsymbol{\hat{\theta}}) & = & \mathbb{E} \{ [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})] [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})]^{T} \} \\ -& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}]^{T} \} \\ -% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} % \\ -% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\theta} \boldsymbol{\theta}^T \\ -& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, \end{eqnarray*} $$

    where we have used that \( \mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = -\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + -\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\theta}) = \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} \), one obtains an estimate of the variance of the estimate of the \( j \)-th regression coefficient: -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to construct a confidence interval for the estimates.

    In a similar way, we can obtain analytical expressions for say the -expectation values of the parameters \( \boldsymbol{\beta} \) and their variance +expectation values of the parameters \( \boldsymbol{\theta} \) and their variance when we employ Ridge regression, allowing us again to define a confidence interval.

    It is rather straightforward to show that

    $$ -\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +\mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\theta}^{\mathrm{OLS}}. $$

    We see clearly that -\( \mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased. +\( \mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\theta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased.

    We can also compute the variance as

    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, $$ -

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\beta} \) goes to zero.

    +

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\theta} \) goes to zero.

    With this, we can compute the difference

    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\theta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. $$

    The difference is non-negative definite since each component of the matrix product is non-negative definite. -This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\beta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. +This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\theta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.

    diff --git a/doc/pub/week38/html/._week38-bs007.html b/doc/pub/week38/html/._week38-bs007.html index 45a7dcfd9..f10456640 100644 --- a/doc/pub/week38/html/._week38-bs007.html +++ b/doc/pub/week38/html/._week38-bs007.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -261,15 +261,15 @@ distribution with zero mean value and an undetermined variance

    We found above that the outputs \( \boldsymbol{y} \) have a mean value given by -\( \boldsymbol{X}\hat{\boldsymbol{\beta}} \) and variance \( \sigma^2 \). Since the entries to +\( \boldsymbol{X}\hat{\boldsymbol{\theta}} \) and variance \( \sigma^2 \). Since the entries to the design matrix are not stochastic variables, we can assume that the probability distribution of our targets is also a normal distribution -but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\beta}} \). This means that a +but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\theta}} \). This means that a single output \( y_i \) is given by the Gaussian distribution

    $$ -y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\beta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\theta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$ diff --git a/doc/pub/week38/html/._week38-bs008.html b/doc/pub/week38/html/._week38-bs008.html index a9b8a2f52..f0c7d6286 100644 --- a/doc/pub/week38/html/._week38-bs008.html +++ b/doc/pub/week38/html/._week38-bs008.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -257,15 +257,15 @@ MathJax.Hub.Config({ We define this distribution as

    $$ -p(y_i, \boldsymbol{X}\vert\boldsymbol{\beta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}, +p(y_i, \boldsymbol{X}\vert\boldsymbol{\theta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}, $$ -

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\beta} \).

    +

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\theta} \).

    Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event \( \boldsymbol{y} \) as the product of the single events, that is we have

    $$ -p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta}). +p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta}). $$

    We will write this in a more compact form reserving \( \boldsymbol{D} \) for the domain of events, including the ouputs (targets) and the inputs. That is @@ -279,10 +279,10 @@ $$ We can now rewrite the above probability as

    $$ -p(\boldsymbol{D}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +p(\boldsymbol{D}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$ -

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\beta} \).

    +

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\theta} \).

    diff --git a/doc/pub/week38/html/._week38-bs009.html b/doc/pub/week38/html/._week38-bs009.html index 05c5ab4a5..43a34c58b 100644 --- a/doc/pub/week38/html/._week38-bs009.html +++ b/doc/pub/week38/html/._week38-bs009.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -261,7 +261,7 @@ data is the most probable.

    We will assume here that our events are given by the above Gaussian -distribution and we will determine the optimal parameters \( \beta \) by +distribution and we will determine the optimal parameters \( \theta \) by maximizing the above PDF. However, computing the derivatives of a product function is cumbersome and can easily lead to overflow and/or underflowproblems, with potentials for loss of numerical precision. diff --git a/doc/pub/week38/html/._week38-bs010.html b/doc/pub/week38/html/._week38-bs010.html index 923142ef3..9db7b117c 100644 --- a/doc/pub/week38/html/._week38-bs010.html +++ b/doc/pub/week38/html/._week38-bs010.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -256,23 +256,23 @@ MathJax.Hub.Config({

    We could now define a new cost function to minimize, namely the negative logarithm of the above PDF

    $$ -C(\boldsymbol{\beta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}, +C(\boldsymbol{\theta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}, $$

    which becomes

    $$ -C(\boldsymbol{\beta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\vert\vert_2^2}{2\sigma^2}. +C(\boldsymbol{\theta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})\vert\vert_2^2}{2\sigma^2}. $$ -

    Taking the derivative of the new cost function with respect to the parameters \( \beta \) we recognize our familiar OLS equation, namely

    +

    Taking the derivative of the new cost function with respect to the parameters \( \theta \) we recognize our familiar OLS equation, namely

    $$ -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right) =0, +\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta}\right) =0, $$ -

    which leads to the well-known OLS equation for the optimal paramters \( \beta \)

    +

    which leads to the well-known OLS equation for the optimal paramters \( \theta \)

    $$ -\hat{\boldsymbol{\beta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! +\hat{\boldsymbol{\theta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! $$

    Next week we will make a similar analysis for Ridge and Lasso regression

    diff --git a/doc/pub/week38/html/._week38-bs011.html b/doc/pub/week38/html/._week38-bs011.html index b2e8d42b9..bc24378e0 100644 --- a/doc/pub/week38/html/._week38-bs011.html +++ b/doc/pub/week38/html/._week38-bs011.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs012.html b/doc/pub/week38/html/._week38-bs012.html index f5c627683..95e64abfa 100644 --- a/doc/pub/week38/html/._week38-bs012.html +++ b/doc/pub/week38/html/._week38-bs012.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs013.html b/doc/pub/week38/html/._week38-bs013.html index be9a22264..f89833211 100644 --- a/doc/pub/week38/html/._week38-bs013.html +++ b/doc/pub/week38/html/._week38-bs013.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs014.html b/doc/pub/week38/html/._week38-bs014.html index bf9314ac2..91edf5963 100644 --- a/doc/pub/week38/html/._week38-bs014.html +++ b/doc/pub/week38/html/._week38-bs014.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs015.html b/doc/pub/week38/html/._week38-bs015.html index 79cf8e0f3..7edc59cb4 100644 --- a/doc/pub/week38/html/._week38-bs015.html +++ b/doc/pub/week38/html/._week38-bs015.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs016.html b/doc/pub/week38/html/._week38-bs016.html index 4e4dc6fe0..35a35b345 100644 --- a/doc/pub/week38/html/._week38-bs016.html +++ b/doc/pub/week38/html/._week38-bs016.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs017.html b/doc/pub/week38/html/._week38-bs017.html index 9dc69805a..c8ac26ca6 100644 --- a/doc/pub/week38/html/._week38-bs017.html +++ b/doc/pub/week38/html/._week38-bs017.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs018.html b/doc/pub/week38/html/._week38-bs018.html index 73ed302ec..11422bf7b 100644 --- a/doc/pub/week38/html/._week38-bs018.html +++ b/doc/pub/week38/html/._week38-bs018.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs019.html b/doc/pub/week38/html/._week38-bs019.html index 37eb3b102..5e0dbe7ed 100644 --- a/doc/pub/week38/html/._week38-bs019.html +++ b/doc/pub/week38/html/._week38-bs019.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs020.html b/doc/pub/week38/html/._week38-bs020.html index f3c96de7b..c12385909 100644 --- a/doc/pub/week38/html/._week38-bs020.html +++ b/doc/pub/week38/html/._week38-bs020.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs021.html b/doc/pub/week38/html/._week38-bs021.html index 2d5ec0a71..d183fc22b 100644 --- a/doc/pub/week38/html/._week38-bs021.html +++ b/doc/pub/week38/html/._week38-bs021.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs022.html b/doc/pub/week38/html/._week38-bs022.html index 46d086436..39b938f7b 100644 --- a/doc/pub/week38/html/._week38-bs022.html +++ b/doc/pub/week38/html/._week38-bs022.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs023.html b/doc/pub/week38/html/._week38-bs023.html index 8c2b83067..26919cfd8 100644 --- a/doc/pub/week38/html/._week38-bs023.html +++ b/doc/pub/week38/html/._week38-bs023.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -255,15 +255,15 @@ MathJax.Hub.Config({

    Confidence intervals are used in statistics and represent a type of estimate computed from the observed data. This gives a range of values for an -unknown parameter such as the parameters \( \boldsymbol{\beta} \) from linear regression. +unknown parameter such as the parameters \( \boldsymbol{\theta} \) from linear regression.

    -

    With the OLS expressions for the parameters \( \boldsymbol{\beta} \) we found -\( \mathbb{E}(\boldsymbol{\beta}) = \boldsymbol{\beta} \), which means that the estimator of the regression parameters is unbiased. +

    With the OLS expressions for the parameters \( \boldsymbol{\theta} \) we found +\( \mathbb{E}(\boldsymbol{\theta}) = \boldsymbol{\theta} \), which means that the estimator of the regression parameters is unbiased.

    In the exercises this week we show that the variance of the estimate of the \( j \)-th regression coefficient is -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \).

    This quantity can be used to diff --git a/doc/pub/week38/html/._week38-bs024.html b/doc/pub/week38/html/._week38-bs024.html index 1beb11850..c03589561 100644 --- a/doc/pub/week38/html/._week38-bs024.html +++ b/doc/pub/week38/html/._week38-bs024.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -253,14 +253,14 @@ MathJax.Hub.Config({

    Standard Approach based on the Normal Distribution

    -

    We will assume that the parameters \( \beta \) follow a normal +

    We will assume that the parameters \( \theta \) follow a normal distribution. We can then define the confidence interval. Here we will be using as -shorthands \( \mu_{\beta} \) for the above mean value and \( \sigma_{\beta} \) +shorthands \( \mu_{\theta} \) for the above mean value and \( \sigma_{\theta} \) for the standard deviation. We have then a confidence interval

    $$ -\left(\mu_{\beta}\pm \frac{z\sigma_{\beta}}{\sqrt{n}}\right), +\left(\mu_{\theta}\pm \frac{z\sigma_{\theta}}{\sqrt{n}}\right), $$

    where \( z \) defines the level of certainty (or confidence). For a normal diff --git a/doc/pub/week38/html/._week38-bs025.html b/doc/pub/week38/html/._week38-bs025.html index ce808117c..c358a1709 100644 --- a/doc/pub/week38/html/._week38-bs025.html +++ b/doc/pub/week38/html/._week38-bs025.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -253,11 +253,11 @@ MathJax.Hub.Config({

    Resampling methods: Bootstrap background

    -

    Since \( \widehat{\beta} = \widehat{\beta}(\boldsymbol{X}) \) is a function of random variables, -\( \widehat{\beta} \) itself must be a random variable. Thus it has +

    Since \( \widehat{\theta} = \widehat{\theta}(\boldsymbol{X}) \) is a function of random variables, +\( \widehat{\theta} \) itself must be a random variable. Thus it has a pdf, call this function \( p(\boldsymbol{t}) \). The aim of the bootstrap is to estimate \( p(\boldsymbol{t}) \) by the relative frequency of -\( \widehat{\beta} \). You can think of this as using a histogram +\( \widehat{\theta} \). You can think of this as using a histogram in the place of \( p(\boldsymbol{t}) \). If the relative frequency closely resembles \( p(\vec{t}) \), then using numerics, it is straight forward to estimate all the interesting parameters of \( p(\boldsymbol{t}) \) using point diff --git a/doc/pub/week38/html/._week38-bs026.html b/doc/pub/week38/html/._week38-bs026.html index f65e1fd4e..343b628bb 100644 --- a/doc/pub/week38/html/._week38-bs026.html +++ b/doc/pub/week38/html/._week38-bs026.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -253,7 +253,7 @@ MathJax.Hub.Config({

    Resampling methods: More Bootstrap background

    -

    In the case that \( \widehat{\beta} \) has +

    In the case that \( \widehat{\theta} \) has more than one component, and the components are independent, we use the same estimator on each component separately. If the probability density function of \( X_i \), \( p(x) \), had been known, then it would have @@ -261,11 +261,11 @@ been straightforward to do this by:

    1. Drawing lots of numbers from \( p(x) \), suppose we call one such set of numbers \( (X_1^*, X_2^*, \cdots, X_n^*) \).
    2. -
    3. Then using these numbers, we could compute a replica of \( \widehat{\beta} \) called \( \widehat{\beta}^* \).
    4. +
    5. Then using these numbers, we could compute a replica of \( \widehat{\theta} \) called \( \widehat{\theta}^* \).

    By repeated use of the above two points, many -estimates of \( \widehat{\beta} \) can be obtained. The -idea is to use the relative frequency of \( \widehat{\beta}^* \) +estimates of \( \widehat{\theta} \) can be obtained. The +idea is to use the relative frequency of \( \widehat{\theta}^* \) (think of a histogram) as an estimate of \( p(\boldsymbol{t}) \).

    diff --git a/doc/pub/week38/html/._week38-bs027.html b/doc/pub/week38/html/._week38-bs027.html index a82e7c109..252f2828d 100644 --- a/doc/pub/week38/html/._week38-bs027.html +++ b/doc/pub/week38/html/._week38-bs027.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs028.html b/doc/pub/week38/html/._week38-bs028.html index c99c39b20..0883f9f43 100644 --- a/doc/pub/week38/html/._week38-bs028.html +++ b/doc/pub/week38/html/._week38-bs028.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -258,18 +258,18 @@ MathJax.Hub.Config({
    1. Draw with replacement \( n \) numbers for the observed variables \( \boldsymbol{x} = (x_1,x_2,\cdots,x_n) \).
    2. Define a vector \( \boldsymbol{x}^* \) containing the values which were drawn from \( \boldsymbol{x} \).
    3. -
    4. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\beta}^* \) by evaluating \( \widehat \beta \) under the observations \( \boldsymbol{x}^* \).
    5. +
    6. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\theta}^* \) by evaluating \( \widehat \theta \) under the observations \( \boldsymbol{x}^* \).
    7. Repeat this process \( k \) times.

    When you are done, you can draw a histogram of the relative frequency -of \( \widehat \beta^* \). This is your estimate of the probability +of \( \widehat \theta^* \). This is your estimate of the probability distribution \( p(t) \). Using this probability distribution you can estimate any statistics thereof. In principle you never draw the -histogram of the relative frequency of \( \widehat{\beta}^* \). Instead +histogram of the relative frequency of \( \widehat{\theta}^* \). Instead you use the estimators corresponding to the statistic of interest. For example, if you are interested in estimating the variance of \( \widehat -\beta \), apply the etsimator \( \widehat \sigma^2 \) to the values -\( \widehat \beta^* \). +\theta \), apply the etsimator \( \widehat \sigma^2 \) to the values +\( \widehat \theta^* \).

    diff --git a/doc/pub/week38/html/._week38-bs029.html b/doc/pub/week38/html/._week38-bs029.html index d80a351d7..25ed7c5d6 100644 --- a/doc/pub/week38/html/._week38-bs029.html +++ b/doc/pub/week38/html/._week38-bs029.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({

  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs030.html b/doc/pub/week38/html/._week38-bs030.html index ae9659eeb..14270e4da 100644 --- a/doc/pub/week38/html/._week38-bs030.html +++ b/doc/pub/week38/html/._week38-bs030.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs031.html b/doc/pub/week38/html/._week38-bs031.html index fe761b456..dc864c65f 100644 --- a/doc/pub/week38/html/._week38-bs031.html +++ b/doc/pub/week38/html/._week38-bs031.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • @@ -270,13 +270,13 @@ $$

    In our derivation of the ordinary least squares method we defined then an approximation to the function \( f \) in terms of the parameters -\( \boldsymbol{\beta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, -that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta} \). +\( \boldsymbol{\theta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, +that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\theta} \).

    -

    Thereafter we found the parameters \( \boldsymbol{\beta} \) by optimizing the means squared error via the so-called cost function

    +

    Thereafter we found the parameters \( \boldsymbol{\theta} \) by optimizing the means squared error via the so-called cost function

    $$ -C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +C(\boldsymbol{X},\boldsymbol{\theta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. $$

    We can rewrite this as

    diff --git a/doc/pub/week38/html/._week38-bs032.html b/doc/pub/week38/html/._week38-bs032.html index acff1d63e..a756abf4e 100644 --- a/doc/pub/week38/html/._week38-bs032.html +++ b/doc/pub/week38/html/._week38-bs032.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs033.html b/doc/pub/week38/html/._week38-bs033.html index a7fab434a..040e257e0 100644 --- a/doc/pub/week38/html/._week38-bs033.html +++ b/doc/pub/week38/html/._week38-bs033.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs034.html b/doc/pub/week38/html/._week38-bs034.html index 8f2b4e765..ab92540b5 100644 --- a/doc/pub/week38/html/._week38-bs034.html +++ b/doc/pub/week38/html/._week38-bs034.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs035.html b/doc/pub/week38/html/._week38-bs035.html index 063d9d50b..c015c57ee 100644 --- a/doc/pub/week38/html/._week38-bs035.html +++ b/doc/pub/week38/html/._week38-bs035.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs036.html b/doc/pub/week38/html/._week38-bs036.html index 1000de440..43db9e657 100644 --- a/doc/pub/week38/html/._week38-bs036.html +++ b/doc/pub/week38/html/._week38-bs036.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs037.html b/doc/pub/week38/html/._week38-bs037.html index 17c7acd86..0a8a8f285 100644 --- a/doc/pub/week38/html/._week38-bs037.html +++ b/doc/pub/week38/html/._week38-bs037.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs038.html b/doc/pub/week38/html/._week38-bs038.html index d71c27c30..64d21ddac 100644 --- a/doc/pub/week38/html/._week38-bs038.html +++ b/doc/pub/week38/html/._week38-bs038.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs039.html b/doc/pub/week38/html/._week38-bs039.html index 4f07ef8da..1b0205ed8 100644 --- a/doc/pub/week38/html/._week38-bs039.html +++ b/doc/pub/week38/html/._week38-bs039.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs040.html b/doc/pub/week38/html/._week38-bs040.html index 8410ce312..f379528bb 100644 --- a/doc/pub/week38/html/._week38-bs040.html +++ b/doc/pub/week38/html/._week38-bs040.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs041.html b/doc/pub/week38/html/._week38-bs041.html index 20ae1dda2..a3959bb0b 100644 --- a/doc/pub/week38/html/._week38-bs041.html +++ b/doc/pub/week38/html/._week38-bs041.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/._week38-bs042.html b/doc/pub/week38/html/._week38-bs042.html index dfa972b94..ba0ebeb22 100644 --- a/doc/pub/week38/html/._week38-bs042.html +++ b/doc/pub/week38/html/._week38-bs042.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/week38-bs.html b/doc/pub/week38/html/week38-bs.html index b468e462d..0c055159c 100644 --- a/doc/pub/week38/html/week38-bs.html +++ b/doc/pub/week38/html/week38-bs.html @@ -51,10 +51,10 @@ doconce format html week38.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -203,7 +203,7 @@ MathJax.Hub.Config({
  • Linking the regression analysis with a statistical interpretation
  • Assumptions made
  • Expectation value and variance
  • -
  • Expectation value and variance for \( \boldsymbol{\beta} \)
  • +
  • Expectation value and variance for \( \boldsymbol{\theta} \)
  • Deriving OLS from a probability distribution
  • Independent and Identically Distributed (iid)
  • Maximum Likelihood Estimation (MLE)
  • diff --git a/doc/pub/week38/html/week38-reveal.html b/doc/pub/week38/html/week38-reveal.html index 8775b77ae..9ec741115 100644 --- a/doc/pub/week38/html/week38-reveal.html +++ b/doc/pub/week38/html/week38-reveal.html @@ -233,7 +233,7 @@ move from a linear algebra analysis to a statistical analysis. In particular, we will focus on what the regularization terms can result in. We will amongst other things show that the regularization parameter can reduce considerably the variance of the parameters -\( \beta \). +\( \theta \).

    On of the advantages of doing linear regression is that we actually end up with @@ -260,7 +260,7 @@ $$

    The randomness of \( \varepsilon_i \) implies that \( \mathbf{y}_i \) is also a random variable. In particular, \( \mathbf{y}_i \) is normally distributed, because \( \varepsilon_i \sim -\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\beta} \) is a +\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\theta} \) is a non-random scalar. To specify the parameters of the distribution of \( \mathbf{y}_i \) we need to calculate its first two moments.

    @@ -289,7 +289,7 @@ function \( f \) is approximated by \( \boldsymbol{\tilde{y}} \) where we want t

     
    $$ -\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\theta}. $$

     

    @@ -302,8 +302,8 @@ $$ $$ \begin{align*} \mathbb{E}(y_i) & = -\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) -\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\theta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \theta, \end{align*} $$

     
    @@ -316,30 +316,30 @@ $$ \begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i - \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - [\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, -\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & -= \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i -\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, -\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 -\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + -\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\theta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \theta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, \mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. \end{align*} $$

     
    -

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with -mean value \( \boldsymbol{X}\boldsymbol{\beta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD). +

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with +mean value \( \boldsymbol{X}\boldsymbol{\theta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD).

    -

    Expectation value and variance for \( \boldsymbol{\beta} \)

    +

    Expectation value and variance for \( \boldsymbol{\theta} \)

    -

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\beta}} \) we can evaluate the expectation value

    +

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\theta}} \) we can evaluate the expectation value

     
    $$ -\mathbb{E}(\boldsymbol{\hat{\beta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +\mathbb{E}(\boldsymbol{\hat{\theta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\theta}=\boldsymbol{\theta}. $$

     
    @@ -347,78 +347,78 @@ $$

    We can also calculate the variance

    -

    The variance of the optimal value \( \boldsymbol{\hat{\beta}} \) is

    +

    The variance of the optimal value \( \boldsymbol{\hat{\theta}} \) is

     
    $$ \begin{eqnarray*} -\mbox{Var}(\boldsymbol{\hat{\beta}}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\mbox{Var}(\boldsymbol{\hat{\theta}}) & = & \mathbb{E} \{ [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})] [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})]^{T} \} \\ -& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}]^{T} \} \\ -% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} % \\ -% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\theta} \boldsymbol{\theta}^T \\ -& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, \end{eqnarray*} $$

     

    where we have used that \( \mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = -\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + -\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\theta}) = \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} \), one obtains an estimate of the variance of the estimate of the \( j \)-th regression coefficient: -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to construct a confidence interval for the estimates.

    In a similar way, we can obtain analytical expressions for say the -expectation values of the parameters \( \boldsymbol{\beta} \) and their variance +expectation values of the parameters \( \boldsymbol{\theta} \) and their variance when we employ Ridge regression, allowing us again to define a confidence interval.

    It is rather straightforward to show that

     
    $$ -\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +\mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\theta}^{\mathrm{OLS}}. $$

     

    We see clearly that -\( \mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased. +\( \mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\theta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased.

    We can also compute the variance as

     
    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, $$

     
    -

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\beta} \) goes to zero.

    +

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\theta} \) goes to zero.

    With this, we can compute the difference

     
    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\theta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. $$

     

    The difference is non-negative definite since each component of the matrix product is non-negative definite. -This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\beta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. +This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\theta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.

    @@ -433,16 +433,16 @@ distribution with zero mean value and an undetermined variance

    We found above that the outputs \( \boldsymbol{y} \) have a mean value given by -\( \boldsymbol{X}\hat{\boldsymbol{\beta}} \) and variance \( \sigma^2 \). Since the entries to +\( \boldsymbol{X}\hat{\boldsymbol{\theta}} \) and variance \( \sigma^2 \). Since the entries to the design matrix are not stochastic variables, we can assume that the probability distribution of our targets is also a normal distribution -but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\beta}} \). This means that a +but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\theta}} \). This means that a single output \( y_i \) is given by the Gaussian distribution

     
    $$ -y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\beta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\theta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$

     
    @@ -455,17 +455,17 @@ We define this distribution as

     
    $$ -p(y_i, \boldsymbol{X}\vert\boldsymbol{\beta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}, +p(y_i, \boldsymbol{X}\vert\boldsymbol{\theta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}, $$

     
    -

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\beta} \).

    +

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\theta} \).

    Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event \( \boldsymbol{y} \) as the product of the single events, that is we have

     
    $$ -p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta}). +p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta}). $$

     
    @@ -483,11 +483,11 @@ We can now rewrite the above probability as

     
    $$ -p(\boldsymbol{D}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +p(\boldsymbol{D}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$

     
    -

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\beta} \).

    +

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\theta} \).

    @@ -501,7 +501,7 @@ data is the most probable.

    We will assume here that our events are given by the above Gaussian -distribution and we will determine the optimal parameters \( \beta \) by +distribution and we will determine the optimal parameters \( \theta \) by maximizing the above PDF. However, computing the derivatives of a product function is cumbersome and can easily lead to overflow and/or underflowproblems, with potentials for loss of numerical precision. @@ -526,29 +526,29 @@ is equivalent to the maximization/minimization of the function itself.

     
    $$ -C(\boldsymbol{\beta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}, +C(\boldsymbol{\theta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}, $$

     

    which becomes

     
    $$ -C(\boldsymbol{\beta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\vert\vert_2^2}{2\sigma^2}. +C(\boldsymbol{\theta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})\vert\vert_2^2}{2\sigma^2}. $$

     
    -

    Taking the derivative of the new cost function with respect to the parameters \( \beta \) we recognize our familiar OLS equation, namely

    +

    Taking the derivative of the new cost function with respect to the parameters \( \theta \) we recognize our familiar OLS equation, namely

     
    $$ -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right) =0, +\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta}\right) =0, $$

     
    -

    which leads to the well-known OLS equation for the optimal paramters \( \beta \)

    +

    which leads to the well-known OLS equation for the optimal paramters \( \theta \)

     
    $$ -\hat{\boldsymbol{\beta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! +\hat{\boldsymbol{\theta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! $$

     
    @@ -879,15 +879,15 @@ finite \( m \), it is not always possible to find a closed form /analytic expres

    Confidence intervals are used in statistics and represent a type of estimate computed from the observed data. This gives a range of values for an -unknown parameter such as the parameters \( \boldsymbol{\beta} \) from linear regression. +unknown parameter such as the parameters \( \boldsymbol{\theta} \) from linear regression.

    -

    With the OLS expressions for the parameters \( \boldsymbol{\beta} \) we found -\( \mathbb{E}(\boldsymbol{\beta}) = \boldsymbol{\beta} \), which means that the estimator of the regression parameters is unbiased. +

    With the OLS expressions for the parameters \( \boldsymbol{\theta} \) we found +\( \mathbb{E}(\boldsymbol{\theta}) = \boldsymbol{\theta} \), which means that the estimator of the regression parameters is unbiased.

    In the exercises this week we show that the variance of the estimate of the \( j \)-th regression coefficient is -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \).

    This quantity can be used to @@ -898,15 +898,15 @@ construct a confidence interval for the estimates.

    Standard Approach based on the Normal Distribution

    -

    We will assume that the parameters \( \beta \) follow a normal +

    We will assume that the parameters \( \theta \) follow a normal distribution. We can then define the confidence interval. Here we will be using as -shorthands \( \mu_{\beta} \) for the above mean value and \( \sigma_{\beta} \) +shorthands \( \mu_{\theta} \) for the above mean value and \( \sigma_{\theta} \) for the standard deviation. We have then a confidence interval

     
    $$ -\left(\mu_{\beta}\pm \frac{z\sigma_{\beta}}{\sqrt{n}}\right), +\left(\mu_{\theta}\pm \frac{z\sigma_{\theta}}{\sqrt{n}}\right), $$

     
    @@ -928,11 +928,11 @@ Bootstrap method, why it works and various theorems related to it.

    Resampling methods: Bootstrap background

    -

    Since \( \widehat{\beta} = \widehat{\beta}(\boldsymbol{X}) \) is a function of random variables, -\( \widehat{\beta} \) itself must be a random variable. Thus it has +

    Since \( \widehat{\theta} = \widehat{\theta}(\boldsymbol{X}) \) is a function of random variables, +\( \widehat{\theta} \) itself must be a random variable. Thus it has a pdf, call this function \( p(\boldsymbol{t}) \). The aim of the bootstrap is to estimate \( p(\boldsymbol{t}) \) by the relative frequency of -\( \widehat{\beta} \). You can think of this as using a histogram +\( \widehat{\theta} \). You can think of this as using a histogram in the place of \( p(\boldsymbol{t}) \). If the relative frequency closely resembles \( p(\vec{t}) \), then using numerics, it is straight forward to estimate all the interesting parameters of \( p(\boldsymbol{t}) \) using point @@ -943,7 +943,7 @@ estimators.

    Resampling methods: More Bootstrap background

    -

    In the case that \( \widehat{\beta} \) has +

    In the case that \( \widehat{\theta} \) has more than one component, and the components are independent, we use the same estimator on each component separately. If the probability density function of \( X_i \), \( p(x) \), had been known, then it would have @@ -951,12 +951,12 @@ been straightforward to do this by:

    1. Drawing lots of numbers from \( p(x) \), suppose we call one such set of numbers \( (X_1^*, X_2^*, \cdots, X_n^*) \).
    2. -

    3. Then using these numbers, we could compute a replica of \( \widehat{\beta} \) called \( \widehat{\beta}^* \).
    4. +

    5. Then using these numbers, we could compute a replica of \( \widehat{\theta} \) called \( \widehat{\theta}^* \).

    By repeated use of the above two points, many -estimates of \( \widehat{\beta} \) can be obtained. The -idea is to use the relative frequency of \( \widehat{\beta}^* \) +estimates of \( \widehat{\theta} \) can be obtained. The +idea is to use the relative frequency of \( \widehat{\theta}^* \) (think of a histogram) as an estimate of \( p(\boldsymbol{t}) \).

    @@ -986,19 +986,19 @@ result in some asymptotic sense? The answer is yes.

    1. Draw with replacement \( n \) numbers for the observed variables \( \boldsymbol{x} = (x_1,x_2,\cdots,x_n) \).
    2. Define a vector \( \boldsymbol{x}^* \) containing the values which were drawn from \( \boldsymbol{x} \).
    3. -

    4. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\beta}^* \) by evaluating \( \widehat \beta \) under the observations \( \boldsymbol{x}^* \).
    5. +

    6. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\theta}^* \) by evaluating \( \widehat \theta \) under the observations \( \boldsymbol{x}^* \).
    7. Repeat this process \( k \) times.

    When you are done, you can draw a histogram of the relative frequency -of \( \widehat \beta^* \). This is your estimate of the probability +of \( \widehat \theta^* \). This is your estimate of the probability distribution \( p(t) \). Using this probability distribution you can estimate any statistics thereof. In principle you never draw the -histogram of the relative frequency of \( \widehat{\beta}^* \). Instead +histogram of the relative frequency of \( \widehat{\theta}^* \). Instead you use the estimators corresponding to the statistic of interest. For example, if you are interested in estimating the variance of \( \widehat -\beta \), apply the etsimator \( \widehat \sigma^2 \) to the values -\( \widehat \beta^* \). +\theta \), apply the etsimator \( \widehat \sigma^2 \) to the values +\( \widehat \theta^* \).

    @@ -1126,14 +1126,14 @@ $$

    In our derivation of the ordinary least squares method we defined then an approximation to the function \( f \) in terms of the parameters -\( \boldsymbol{\beta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, -that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta} \). +\( \boldsymbol{\theta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, +that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\theta} \).

    -

    Thereafter we found the parameters \( \boldsymbol{\beta} \) by optimizing the means squared error via the so-called cost function

    +

    Thereafter we found the parameters \( \boldsymbol{\theta} \) by optimizing the means squared error via the so-called cost function

     
    $$ -C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +C(\boldsymbol{X},\boldsymbol{\theta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. $$

     
    diff --git a/doc/pub/week38/html/week38-solarized.html b/doc/pub/week38/html/week38-solarized.html index 407130d97..de2ecd7f0 100644 --- a/doc/pub/week38/html/week38-solarized.html +++ b/doc/pub/week38/html/week38-solarized.html @@ -78,10 +78,10 @@ div.toc p,a { 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -270,7 +270,7 @@ move from a linear algebra analysis to a statistical analysis. In particular, we will focus on what the regularization terms can result in. We will amongst other things show that the regularization parameter can reduce considerably the variance of the parameters -\( \beta \). +\( \theta \).

    On of the advantages of doing linear regression is that we actually end up with @@ -295,7 +295,7 @@ $$

    The randomness of \( \varepsilon_i \) implies that \( \mathbf{y}_i \) is also a random variable. In particular, \( \mathbf{y}_i \) is normally distributed, because \( \varepsilon_i \sim -\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\beta} \) is a +\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\theta} \) is a non-random scalar. To specify the parameters of the distribution of \( \mathbf{y}_i \) we need to calculate its first two moments.

    @@ -320,7 +320,7 @@ $$ function \( f \) is approximated by \( \boldsymbol{\tilde{y}} \) where we want to minimize \( (\boldsymbol{y}-\boldsymbol{\tilde{y}})^2 \), our MSE, with

    $$ -\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\theta}. $$ @@ -331,8 +331,8 @@ $$ $$ \begin{align*} \mathbb{E}(y_i) & = -\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) -\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\theta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \theta, \end{align*} $$ @@ -343,97 +343,97 @@ $$ \begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i - \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - [\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, -\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & -= \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i -\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, -\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 -\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + -\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\theta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \theta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, \mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. \end{align*} $$ -

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with -mean value \( \boldsymbol{X}\boldsymbol{\beta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD). +

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with +mean value \( \boldsymbol{X}\boldsymbol{\theta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD).











    -

    Expectation value and variance for \( \boldsymbol{\beta} \)

    +

    Expectation value and variance for \( \boldsymbol{\theta} \)

    -

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\beta}} \) we can evaluate the expectation value

    +

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\theta}} \) we can evaluate the expectation value

    $$ -\mathbb{E}(\boldsymbol{\hat{\beta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +\mathbb{E}(\boldsymbol{\hat{\theta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\theta}=\boldsymbol{\theta}. $$

    This means that the estimator of the regression parameters is unbiased.

    We can also calculate the variance

    -

    The variance of the optimal value \( \boldsymbol{\hat{\beta}} \) is

    +

    The variance of the optimal value \( \boldsymbol{\hat{\theta}} \) is

    $$ \begin{eqnarray*} -\mbox{Var}(\boldsymbol{\hat{\beta}}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\mbox{Var}(\boldsymbol{\hat{\theta}}) & = & \mathbb{E} \{ [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})] [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})]^{T} \} \\ -& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}]^{T} \} \\ -% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} % \\ -% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\theta} \boldsymbol{\theta}^T \\ -& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, \end{eqnarray*} $$

    where we have used that \( \mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = -\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + -\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\theta}) = \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} \), one obtains an estimate of the variance of the estimate of the \( j \)-th regression coefficient: -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to construct a confidence interval for the estimates.

    In a similar way, we can obtain analytical expressions for say the -expectation values of the parameters \( \boldsymbol{\beta} \) and their variance +expectation values of the parameters \( \boldsymbol{\theta} \) and their variance when we employ Ridge regression, allowing us again to define a confidence interval.

    It is rather straightforward to show that

    $$ -\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +\mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\theta}^{\mathrm{OLS}}. $$

    We see clearly that -\( \mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased. +\( \mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\theta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased.

    We can also compute the variance as

    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, $$ -

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\beta} \) goes to zero.

    +

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\theta} \) goes to zero.

    With this, we can compute the difference

    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\theta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. $$

    The difference is non-negative definite since each component of the matrix product is non-negative definite. -This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\beta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. +This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\theta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.











    @@ -447,15 +447,15 @@ distribution with zero mean value and an undetermined variance

    We found above that the outputs \( \boldsymbol{y} \) have a mean value given by -\( \boldsymbol{X}\hat{\boldsymbol{\beta}} \) and variance \( \sigma^2 \). Since the entries to +\( \boldsymbol{X}\hat{\boldsymbol{\theta}} \) and variance \( \sigma^2 \). Since the entries to the design matrix are not stochastic variables, we can assume that the probability distribution of our targets is also a normal distribution -but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\beta}} \). This means that a +but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\theta}} \). This means that a single output \( y_i \) is given by the Gaussian distribution

    $$ -y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\beta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\theta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$ @@ -466,15 +466,15 @@ $$ We define this distribution as

    $$ -p(y_i, \boldsymbol{X}\vert\boldsymbol{\beta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}, +p(y_i, \boldsymbol{X}\vert\boldsymbol{\theta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}, $$ -

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\beta} \).

    +

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\theta} \).

    Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event \( \boldsymbol{y} \) as the product of the single events, that is we have

    $$ -p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta}). +p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta}). $$

    We will write this in a more compact form reserving \( \boldsymbol{D} \) for the domain of events, including the ouputs (targets) and the inputs. That is @@ -488,10 +488,10 @@ $$ We can now rewrite the above probability as

    $$ -p(\boldsymbol{D}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +p(\boldsymbol{D}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$ -

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\beta} \).

    +

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\theta} \).











    Maximum Likelihood Estimation (MLE)

    @@ -504,7 +504,7 @@ data is the most probable.

    We will assume here that our events are given by the above Gaussian -distribution and we will determine the optimal parameters \( \beta \) by +distribution and we will determine the optimal parameters \( \theta \) by maximizing the above PDF. However, computing the derivatives of a product function is cumbersome and can easily lead to overflow and/or underflowproblems, with potentials for loss of numerical precision. @@ -527,23 +527,23 @@ is equivalent to the maximization/minimization of the function itself.

    We could now define a new cost function to minimize, namely the negative logarithm of the above PDF

    $$ -C(\boldsymbol{\beta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}, +C(\boldsymbol{\theta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}, $$

    which becomes

    $$ -C(\boldsymbol{\beta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\vert\vert_2^2}{2\sigma^2}. +C(\boldsymbol{\theta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})\vert\vert_2^2}{2\sigma^2}. $$ -

    Taking the derivative of the new cost function with respect to the parameters \( \beta \) we recognize our familiar OLS equation, namely

    +

    Taking the derivative of the new cost function with respect to the parameters \( \theta \) we recognize our familiar OLS equation, namely

    $$ -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right) =0, +\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta}\right) =0, $$ -

    which leads to the well-known OLS equation for the optimal paramters \( \beta \)

    +

    which leads to the well-known OLS equation for the optimal paramters \( \theta \)

    $$ -\hat{\boldsymbol{\beta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! +\hat{\boldsymbol{\theta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! $$

    Next week we will make a similar analysis for Ridge and Lasso regression

    @@ -838,15 +838,15 @@ finite \( m \), it is not always possible to find a closed form /analytic expres

    Confidence intervals are used in statistics and represent a type of estimate computed from the observed data. This gives a range of values for an -unknown parameter such as the parameters \( \boldsymbol{\beta} \) from linear regression. +unknown parameter such as the parameters \( \boldsymbol{\theta} \) from linear regression.

    -

    With the OLS expressions for the parameters \( \boldsymbol{\beta} \) we found -\( \mathbb{E}(\boldsymbol{\beta}) = \boldsymbol{\beta} \), which means that the estimator of the regression parameters is unbiased. +

    With the OLS expressions for the parameters \( \boldsymbol{\theta} \) we found +\( \mathbb{E}(\boldsymbol{\theta}) = \boldsymbol{\theta} \), which means that the estimator of the regression parameters is unbiased.

    In the exercises this week we show that the variance of the estimate of the \( j \)-th regression coefficient is -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \).

    This quantity can be used to @@ -856,14 +856,14 @@ construct a confidence interval for the estimates.









    Standard Approach based on the Normal Distribution

    -

    We will assume that the parameters \( \beta \) follow a normal +

    We will assume that the parameters \( \theta \) follow a normal distribution. We can then define the confidence interval. Here we will be using as -shorthands \( \mu_{\beta} \) for the above mean value and \( \sigma_{\beta} \) +shorthands \( \mu_{\theta} \) for the above mean value and \( \sigma_{\theta} \) for the standard deviation. We have then a confidence interval

    $$ -\left(\mu_{\beta}\pm \frac{z\sigma_{\beta}}{\sqrt{n}}\right), +\left(\mu_{\theta}\pm \frac{z\sigma_{\theta}}{\sqrt{n}}\right), $$

    where \( z \) defines the level of certainty (or confidence). For a normal @@ -883,11 +883,11 @@ Bootstrap method, why it works and various theorems related to it.









    Resampling methods: Bootstrap background

    -

    Since \( \widehat{\beta} = \widehat{\beta}(\boldsymbol{X}) \) is a function of random variables, -\( \widehat{\beta} \) itself must be a random variable. Thus it has +

    Since \( \widehat{\theta} = \widehat{\theta}(\boldsymbol{X}) \) is a function of random variables, +\( \widehat{\theta} \) itself must be a random variable. Thus it has a pdf, call this function \( p(\boldsymbol{t}) \). The aim of the bootstrap is to estimate \( p(\boldsymbol{t}) \) by the relative frequency of -\( \widehat{\beta} \). You can think of this as using a histogram +\( \widehat{\theta} \). You can think of this as using a histogram in the place of \( p(\boldsymbol{t}) \). If the relative frequency closely resembles \( p(\vec{t}) \), then using numerics, it is straight forward to estimate all the interesting parameters of \( p(\boldsymbol{t}) \) using point @@ -897,7 +897,7 @@ estimators.









    Resampling methods: More Bootstrap background

    -

    In the case that \( \widehat{\beta} \) has +

    In the case that \( \widehat{\theta} \) has more than one component, and the components are independent, we use the same estimator on each component separately. If the probability density function of \( X_i \), \( p(x) \), had been known, then it would have @@ -905,11 +905,11 @@ been straightforward to do this by:

    1. Drawing lots of numbers from \( p(x) \), suppose we call one such set of numbers \( (X_1^*, X_2^*, \cdots, X_n^*) \).
    2. -
    3. Then using these numbers, we could compute a replica of \( \widehat{\beta} \) called \( \widehat{\beta}^* \).
    4. +
    5. Then using these numbers, we could compute a replica of \( \widehat{\theta} \) called \( \widehat{\theta}^* \).

    By repeated use of the above two points, many -estimates of \( \widehat{\beta} \) can be obtained. The -idea is to use the relative frequency of \( \widehat{\beta}^* \) +estimates of \( \widehat{\theta} \) can be obtained. The +idea is to use the relative frequency of \( \widehat{\theta}^* \) (think of a histogram) as an estimate of \( p(\boldsymbol{t}) \).

    @@ -937,18 +937,18 @@ result in some asymptotic sense? The answer is yes.
    1. Draw with replacement \( n \) numbers for the observed variables \( \boldsymbol{x} = (x_1,x_2,\cdots,x_n) \).
    2. Define a vector \( \boldsymbol{x}^* \) containing the values which were drawn from \( \boldsymbol{x} \).
    3. -
    4. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\beta}^* \) by evaluating \( \widehat \beta \) under the observations \( \boldsymbol{x}^* \).
    5. +
    6. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\theta}^* \) by evaluating \( \widehat \theta \) under the observations \( \boldsymbol{x}^* \).
    7. Repeat this process \( k \) times.

    When you are done, you can draw a histogram of the relative frequency -of \( \widehat \beta^* \). This is your estimate of the probability +of \( \widehat \theta^* \). This is your estimate of the probability distribution \( p(t) \). Using this probability distribution you can estimate any statistics thereof. In principle you never draw the -histogram of the relative frequency of \( \widehat{\beta}^* \). Instead +histogram of the relative frequency of \( \widehat{\theta}^* \). Instead you use the estimators corresponding to the statistic of interest. For example, if you are interested in estimating the variance of \( \widehat -\beta \), apply the etsimator \( \widehat \sigma^2 \) to the values -\( \widehat \beta^* \). +\theta \), apply the etsimator \( \widehat \sigma^2 \) to the values +\( \widehat \theta^* \).











    @@ -1072,13 +1072,13 @@ $$

    In our derivation of the ordinary least squares method we defined then an approximation to the function \( f \) in terms of the parameters -\( \boldsymbol{\beta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, -that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta} \). +\( \boldsymbol{\theta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, +that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\theta} \).

    -

    Thereafter we found the parameters \( \boldsymbol{\beta} \) by optimizing the means squared error via the so-called cost function

    +

    Thereafter we found the parameters \( \boldsymbol{\theta} \) by optimizing the means squared error via the so-called cost function

    $$ -C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +C(\boldsymbol{X},\boldsymbol{\theta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. $$

    We can rewrite this as

    diff --git a/doc/pub/week38/html/week38.html b/doc/pub/week38/html/week38.html index 7aa76c5a4..366c5f576 100644 --- a/doc/pub/week38/html/week38.html +++ b/doc/pub/week38/html/week38.html @@ -155,10 +155,10 @@ div.toc p,a { 2, None, 'expectation-value-and-variance'), - ('Expectation value and variance for $\\boldsymbol{\\beta}$', + ('Expectation value and variance for $\\boldsymbol{\\theta}$', 2, None, - 'expectation-value-and-variance-for-boldsymbol-beta'), + 'expectation-value-and-variance-for-boldsymbol-theta'), ('Deriving OLS from a probability distribution', 2, None, @@ -347,7 +347,7 @@ move from a linear algebra analysis to a statistical analysis. In particular, we will focus on what the regularization terms can result in. We will amongst other things show that the regularization parameter can reduce considerably the variance of the parameters -\( \beta \). +\( \theta \).

    On of the advantages of doing linear regression is that we actually end up with @@ -372,7 +372,7 @@ $$

    The randomness of \( \varepsilon_i \) implies that \( \mathbf{y}_i \) is also a random variable. In particular, \( \mathbf{y}_i \) is normally distributed, because \( \varepsilon_i \sim -\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\beta} \) is a +\mathcal{N}(0, \sigma^2) \) and \( \mathbf{X}_{i,\ast} \, \boldsymbol{\theta} \) is a non-random scalar. To specify the parameters of the distribution of \( \mathbf{y}_i \) we need to calculate its first two moments.

    @@ -397,7 +397,7 @@ $$ function \( f \) is approximated by \( \boldsymbol{\tilde{y}} \) where we want to minimize \( (\boldsymbol{y}-\boldsymbol{\tilde{y}})^2 \), our MSE, with

    $$ -\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\theta}. $$ @@ -408,8 +408,8 @@ $$ $$ \begin{align*} \mathbb{E}(y_i) & = -\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) -\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\theta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \theta, \end{align*} $$ @@ -420,97 +420,97 @@ $$ \begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i - \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - [\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, -\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & -= \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i -\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, -\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 -\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + -\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\theta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \theta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\theta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta})^2 \\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, \mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. \end{align*} $$ -

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with -mean value \( \boldsymbol{X}\boldsymbol{\beta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD). +

    Hence, \( y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\theta}, \sigma^2) \), that is \( \boldsymbol{y} \) follows a normal distribution with +mean value \( \boldsymbol{X}\boldsymbol{\theta} \) and variance \( \sigma^2 \) (not be confused with the singular values of the SVD).











    -

    Expectation value and variance for \( \boldsymbol{\beta} \)

    +

    Expectation value and variance for \( \boldsymbol{\theta} \)

    -

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\beta}} \) we can evaluate the expectation value

    +

    With the OLS expressions for the optimal parameters \( \boldsymbol{\hat{\theta}} \) we can evaluate the expectation value

    $$ -\mathbb{E}(\boldsymbol{\hat{\beta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +\mathbb{E}(\boldsymbol{\hat{\theta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\theta}=\boldsymbol{\theta}. $$

    This means that the estimator of the regression parameters is unbiased.

    We can also calculate the variance

    -

    The variance of the optimal value \( \boldsymbol{\hat{\beta}} \) is

    +

    The variance of the optimal value \( \boldsymbol{\hat{\theta}} \) is

    $$ \begin{eqnarray*} -\mbox{Var}(\boldsymbol{\hat{\beta}}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\mbox{Var}(\boldsymbol{\hat{\theta}}) & = & \mathbb{E} \{ [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})] [\boldsymbol{\theta} - \mathbb{E}(\boldsymbol{\theta})]^{T} \} \\ -& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\theta}]^{T} \} \\ -% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} % \\ -% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} % \\ -% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\theta} \boldsymbol{\theta}^T \\ -& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +& = & \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, \end{eqnarray*} $$

    where we have used that \( \mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = -\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + -\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\mathbf{X} \, \boldsymbol{\theta} \, \boldsymbol{\theta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn} \). From \( \mbox{Var}(\boldsymbol{\theta}) = \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} \), one obtains an estimate of the variance of the estimate of the \( j \)-th regression coefficient: -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). This may be used to construct a confidence interval for the estimates.

    In a similar way, we can obtain analytical expressions for say the -expectation values of the parameters \( \boldsymbol{\beta} \) and their variance +expectation values of the parameters \( \boldsymbol{\theta} \) and their variance when we employ Ridge regression, allowing us again to define a confidence interval.

    It is rather straightforward to show that

    $$ -\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +\mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\theta}^{\mathrm{OLS}}. $$

    We see clearly that -\( \mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased. +\( \mathbb{E} \big[ \boldsymbol{\theta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\theta}^{\mathrm{OLS}} \) for any \( \lambda > 0 \). We say then that the ridge estimator is biased.

    We can also compute the variance as

    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, $$ -

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\beta} \) goes to zero.

    +

    and it is easy to see that if the parameter \( \lambda \) goes to infinity then the variance of Ridge parameters \( \boldsymbol{\theta} \) goes to zero.

    With this, we can compute the difference

    $$ -\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\mbox{Var}[\boldsymbol{\theta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\theta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. $$

    The difference is non-negative definite since each component of the matrix product is non-negative definite. -This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\beta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. +This means the variance we obtain with the standard OLS will always for \( \lambda > 0 \) be larger than the variance of \( \boldsymbol{\theta} \) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.











    @@ -524,15 +524,15 @@ distribution with zero mean value and an undetermined variance

    We found above that the outputs \( \boldsymbol{y} \) have a mean value given by -\( \boldsymbol{X}\hat{\boldsymbol{\beta}} \) and variance \( \sigma^2 \). Since the entries to +\( \boldsymbol{X}\hat{\boldsymbol{\theta}} \) and variance \( \sigma^2 \). Since the entries to the design matrix are not stochastic variables, we can assume that the probability distribution of our targets is also a normal distribution -but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\beta}} \). This means that a +but now with mean value \( \boldsymbol{X}\hat{\boldsymbol{\theta}} \). This means that a single output \( y_i \) is given by the Gaussian distribution

    $$ -y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\beta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +y_i\sim \mathcal{N}(\boldsymbol{X}_{i,*}\boldsymbol{\theta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$ @@ -543,15 +543,15 @@ $$ We define this distribution as

    $$ -p(y_i, \boldsymbol{X}\vert\boldsymbol{\beta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}, +p(y_i, \boldsymbol{X}\vert\boldsymbol{\theta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}, $$ -

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\beta} \).

    +

    which reads as finding the likelihood of an event \( y_i \) with the input variables \( \boldsymbol{X} \) given the parameters (to be determined) \( \boldsymbol{\theta} \).

    Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event \( \boldsymbol{y} \) as the product of the single events, that is we have

    $$ -p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta}). +p(\boldsymbol{y},\boldsymbol{X}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta}). $$

    We will write this in a more compact form reserving \( \boldsymbol{D} \) for the domain of events, including the ouputs (targets) and the inputs. That is @@ -565,10 +565,10 @@ $$ We can now rewrite the above probability as

    $$ -p(\boldsymbol{D}\vert\boldsymbol{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\beta})^2}{2\sigma^2}\right]}. +p(\boldsymbol{D}\vert\boldsymbol{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\boldsymbol{X}_{i,*}\boldsymbol{\theta})^2}{2\sigma^2}\right]}. $$ -

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\beta} \).

    +

    It is a conditional probability (see below) and reads as the likelihood of a domain of events \( \boldsymbol{D} \) given a set of parameters \( \boldsymbol{\theta} \).











    Maximum Likelihood Estimation (MLE)

    @@ -581,7 +581,7 @@ data is the most probable.

    We will assume here that our events are given by the above Gaussian -distribution and we will determine the optimal parameters \( \beta \) by +distribution and we will determine the optimal parameters \( \theta \) by maximizing the above PDF. However, computing the derivatives of a product function is cumbersome and can easily lead to overflow and/or underflowproblems, with potentials for loss of numerical precision. @@ -604,23 +604,23 @@ is equivalent to the maximization/minimization of the function itself.

    We could now define a new cost function to minimize, namely the negative logarithm of the above PDF

    $$ -C(\boldsymbol{\beta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\beta})}, +C(\boldsymbol{\theta}=-\log{\prod_{i=0}^{n-1}p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\boldsymbol{X}\vert\boldsymbol{\theta})}, $$

    which becomes

    $$ -C(\boldsymbol{\beta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\vert\vert_2^2}{2\sigma^2}. +C(\boldsymbol{\theta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta})\vert\vert_2^2}{2\sigma^2}. $$ -

    Taking the derivative of the new cost function with respect to the parameters \( \beta \) we recognize our familiar OLS equation, namely

    +

    Taking the derivative of the new cost function with respect to the parameters \( \theta \) we recognize our familiar OLS equation, namely

    $$ -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right) =0, +\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\theta}\right) =0, $$ -

    which leads to the well-known OLS equation for the optimal paramters \( \beta \)

    +

    which leads to the well-known OLS equation for the optimal paramters \( \theta \)

    $$ -\hat{\boldsymbol{\beta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! +\hat{\boldsymbol{\theta}}^{\mathrm{OLS}}=\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}! $$

    Next week we will make a similar analysis for Ridge and Lasso regression

    @@ -915,15 +915,15 @@ finite \( m \), it is not always possible to find a closed form /analytic expres

    Confidence intervals are used in statistics and represent a type of estimate computed from the observed data. This gives a range of values for an -unknown parameter such as the parameters \( \boldsymbol{\beta} \) from linear regression. +unknown parameter such as the parameters \( \boldsymbol{\theta} \) from linear regression.

    -

    With the OLS expressions for the parameters \( \boldsymbol{\beta} \) we found -\( \mathbb{E}(\boldsymbol{\beta}) = \boldsymbol{\beta} \), which means that the estimator of the regression parameters is unbiased. +

    With the OLS expressions for the parameters \( \boldsymbol{\theta} \) we found +\( \mathbb{E}(\boldsymbol{\theta}) = \boldsymbol{\theta} \), which means that the estimator of the regression parameters is unbiased.

    In the exercises this week we show that the variance of the estimate of the \( j \)-th regression coefficient is -\( \boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \). +\( \boldsymbol{\sigma}^2 (\boldsymbol{\theta}_j ) = \boldsymbol{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} \).

    This quantity can be used to @@ -933,14 +933,14 @@ construct a confidence interval for the estimates.









    Standard Approach based on the Normal Distribution

    -

    We will assume that the parameters \( \beta \) follow a normal +

    We will assume that the parameters \( \theta \) follow a normal distribution. We can then define the confidence interval. Here we will be using as -shorthands \( \mu_{\beta} \) for the above mean value and \( \sigma_{\beta} \) +shorthands \( \mu_{\theta} \) for the above mean value and \( \sigma_{\theta} \) for the standard deviation. We have then a confidence interval

    $$ -\left(\mu_{\beta}\pm \frac{z\sigma_{\beta}}{\sqrt{n}}\right), +\left(\mu_{\theta}\pm \frac{z\sigma_{\theta}}{\sqrt{n}}\right), $$

    where \( z \) defines the level of certainty (or confidence). For a normal @@ -960,11 +960,11 @@ Bootstrap method, why it works and various theorems related to it.









    Resampling methods: Bootstrap background

    -

    Since \( \widehat{\beta} = \widehat{\beta}(\boldsymbol{X}) \) is a function of random variables, -\( \widehat{\beta} \) itself must be a random variable. Thus it has +

    Since \( \widehat{\theta} = \widehat{\theta}(\boldsymbol{X}) \) is a function of random variables, +\( \widehat{\theta} \) itself must be a random variable. Thus it has a pdf, call this function \( p(\boldsymbol{t}) \). The aim of the bootstrap is to estimate \( p(\boldsymbol{t}) \) by the relative frequency of -\( \widehat{\beta} \). You can think of this as using a histogram +\( \widehat{\theta} \). You can think of this as using a histogram in the place of \( p(\boldsymbol{t}) \). If the relative frequency closely resembles \( p(\vec{t}) \), then using numerics, it is straight forward to estimate all the interesting parameters of \( p(\boldsymbol{t}) \) using point @@ -974,7 +974,7 @@ estimators.









    Resampling methods: More Bootstrap background

    -

    In the case that \( \widehat{\beta} \) has +

    In the case that \( \widehat{\theta} \) has more than one component, and the components are independent, we use the same estimator on each component separately. If the probability density function of \( X_i \), \( p(x) \), had been known, then it would have @@ -982,11 +982,11 @@ been straightforward to do this by:

    1. Drawing lots of numbers from \( p(x) \), suppose we call one such set of numbers \( (X_1^*, X_2^*, \cdots, X_n^*) \).
    2. -
    3. Then using these numbers, we could compute a replica of \( \widehat{\beta} \) called \( \widehat{\beta}^* \).
    4. +
    5. Then using these numbers, we could compute a replica of \( \widehat{\theta} \) called \( \widehat{\theta}^* \).

    By repeated use of the above two points, many -estimates of \( \widehat{\beta} \) can be obtained. The -idea is to use the relative frequency of \( \widehat{\beta}^* \) +estimates of \( \widehat{\theta} \) can be obtained. The +idea is to use the relative frequency of \( \widehat{\theta}^* \) (think of a histogram) as an estimate of \( p(\boldsymbol{t}) \).

    @@ -1014,18 +1014,18 @@ result in some asymptotic sense? The answer is yes.
    1. Draw with replacement \( n \) numbers for the observed variables \( \boldsymbol{x} = (x_1,x_2,\cdots,x_n) \).
    2. Define a vector \( \boldsymbol{x}^* \) containing the values which were drawn from \( \boldsymbol{x} \).
    3. -
    4. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\beta}^* \) by evaluating \( \widehat \beta \) under the observations \( \boldsymbol{x}^* \).
    5. +
    6. Using the vector \( \boldsymbol{x}^* \) compute \( \widehat{\theta}^* \) by evaluating \( \widehat \theta \) under the observations \( \boldsymbol{x}^* \).
    7. Repeat this process \( k \) times.

    When you are done, you can draw a histogram of the relative frequency -of \( \widehat \beta^* \). This is your estimate of the probability +of \( \widehat \theta^* \). This is your estimate of the probability distribution \( p(t) \). Using this probability distribution you can estimate any statistics thereof. In principle you never draw the -histogram of the relative frequency of \( \widehat{\beta}^* \). Instead +histogram of the relative frequency of \( \widehat{\theta}^* \). Instead you use the estimators corresponding to the statistic of interest. For example, if you are interested in estimating the variance of \( \widehat -\beta \), apply the etsimator \( \widehat \sigma^2 \) to the values -\( \widehat \beta^* \). +\theta \), apply the etsimator \( \widehat \sigma^2 \) to the values +\( \widehat \theta^* \).











    @@ -1149,13 +1149,13 @@ $$

    In our derivation of the ordinary least squares method we defined then an approximation to the function \( f \) in terms of the parameters -\( \boldsymbol{\beta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, -that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta} \). +\( \boldsymbol{\theta} \) and the design matrix \( \boldsymbol{X} \) which embody our model, +that is \( \boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\theta} \).

    -

    Thereafter we found the parameters \( \boldsymbol{\beta} \) by optimizing the means squared error via the so-called cost function

    +

    Thereafter we found the parameters \( \boldsymbol{\theta} \) by optimizing the means squared error via the so-called cost function

    $$ -C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +C(\boldsymbol{X},\boldsymbol{\theta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. $$

    We can rewrite this as

    diff --git a/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz b/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz index 98390b475db8b7cce61b56acf1b7472ff30ab6a1..b68a1c64df2b7150b093b04eceb1fc0f255cd041 100644 GIT binary patch delta 63 zcmWN_IT3&`002S$3%>#eC-Fvc6*!=c1!4hPu;r#4Q;zIsD6a4xNGX-n(nu?v{28Q| NK}MNmmgQmH?hhwu4`BcR delta 63 zcmWN_IT3&`002S$3%>#eC(#DQRY;(X1(1LmM8KAt?wIb#dW7uVo*l`hkWwnCrI9~_ Pw9-j0gN!m=%**)!Ql1bv diff --git a/doc/pub/week38/ipynb/week38.ipynb b/doc/pub/week38/ipynb/week38.ipynb index 65d9c7274..544d286d0 100644 --- a/doc/pub/week38/ipynb/week38.ipynb +++ b/doc/pub/week38/ipynb/week38.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "8a0f05f4", + "id": "169923b3", "metadata": { "editable": true }, @@ -14,7 +14,7 @@ }, { "cell_type": "markdown", - "id": "db96d5cf", + "id": "47013ee7", "metadata": { "editable": true }, @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "f2af7718", + "id": "d1fb5464", "metadata": { "editable": true }, @@ -47,7 +47,7 @@ }, { "cell_type": "markdown", - "id": "0caafde2", + "id": "1a34a7ce", "metadata": { "editable": true }, @@ -68,7 +68,7 @@ }, { "cell_type": "markdown", - "id": "84b5e975", + "id": "c6f56f83", "metadata": { "editable": true }, @@ -81,7 +81,7 @@ "particular, we will focus on what the regularization terms can result\n", "in. We will amongst other things show that the regularization\n", "parameter can reduce considerably the variance of the parameters\n", - "$\\beta$.\n", + "$\\theta$.\n", "\n", "On of the advantages of doing linear regression is that we actually end up with\n", "analytical expressions for several statistical quantities. \n", @@ -96,7 +96,7 @@ }, { "cell_type": "markdown", - "id": "45439624", + "id": "3ece0c04", "metadata": { "editable": true }, @@ -112,7 +112,7 @@ }, { "cell_type": "markdown", - "id": "a69df793", + "id": "d36cf6db", "metadata": { "editable": true }, @@ -120,7 +120,7 @@ "The randomness of $\\varepsilon_i$ implies that\n", "$\\mathbf{y}_i$ is also a random variable. In particular,\n", "$\\mathbf{y}_i$ is normally distributed, because $\\varepsilon_i \\sim\n", - "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\beta}$ is a\n", + "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\theta}$ is a\n", "non-random scalar. To specify the parameters of the distribution of\n", "$\\mathbf{y}_i$ we need to calculate its first two moments. \n", "\n", @@ -131,7 +131,7 @@ }, { "cell_type": "markdown", - "id": "1e629d2f", + "id": "5903d2be", "metadata": { "editable": true }, @@ -145,7 +145,7 @@ }, { "cell_type": "markdown", - "id": "30678af7", + "id": "37f4e199", "metadata": { "editable": true }, @@ -157,7 +157,7 @@ }, { "cell_type": "markdown", - "id": "695d00e6", + "id": "7a7b084f", "metadata": { "editable": true }, @@ -168,19 +168,19 @@ }, { "cell_type": "markdown", - "id": "10ccf8a0", + "id": "1ec511a1", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta}.\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "0f5247e3", + "id": "ffe76935", "metadata": { "editable": true }, @@ -192,7 +192,7 @@ }, { "cell_type": "markdown", - "id": "938ff96e", + "id": "e569274a", "metadata": { "editable": true }, @@ -200,15 +200,15 @@ "$$\n", "\\begin{align*} \n", "\\mathbb{E}(y_i) & =\n", - "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}) + \\mathbb{E}(\\varepsilon_i)\n", - "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\beta, \n", + "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}) + \\mathbb{E}(\\varepsilon_i)\n", + "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\theta, \n", "\\end{align*}\n", "$$" ] }, { "cell_type": "markdown", - "id": "92d79165", + "id": "c4dd2623", "metadata": { "editable": true }, @@ -219,7 +219,7 @@ }, { "cell_type": "markdown", - "id": "945a48fa", + "id": "df2f8936", "metadata": { "editable": true }, @@ -228,12 +228,12 @@ "\\begin{align*} \\mbox{Var}(y_i) & = \\mathbb{E} \\{ [y_i\n", "- \\mathbb{E}(y_i)]^2 \\} \\, \\, \\, = \\, \\, \\, \\mathbb{E} ( y_i^2 ) -\n", "[\\mathbb{E}(y_i)]^2 \\\\ & = \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\,\n", - "\\beta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \\\\ &\n", - "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2 \\varepsilon_i\n", - "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", - "\\ast} \\, \\beta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2\n", - "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} +\n", - "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \n", + "\\theta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \\\\ &\n", + "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2 \\varepsilon_i\n", + "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", + "\\ast} \\, \\theta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 + 2\n", + "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta} +\n", + "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta})^2 \n", "\\\\ & = \\mathbb{E}(\\varepsilon_i^2 ) \\, \\, \\, = \\, \\, \\,\n", "\\mbox{Var}(\\varepsilon_i) \\, \\, \\, = \\, \\, \\, \\sigma^2. \n", "\\end{align*}\n", @@ -242,42 +242,42 @@ }, { "cell_type": "markdown", - "id": "8e0a0453", + "id": "0a3e5956", "metadata": { "editable": true }, "source": [ - "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", - "mean value $\\boldsymbol{X}\\boldsymbol{\\beta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." + "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\theta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", + "mean value $\\boldsymbol{X}\\boldsymbol{\\theta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD)." ] }, { "cell_type": "markdown", - "id": "f7f389f4", + "id": "973e45a3", "metadata": { "editable": true }, "source": [ - "## Expectation value and variance for $\\boldsymbol{\\beta}$\n", + "## Expectation value and variance for $\\boldsymbol{\\theta}$\n", "\n", - "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\beta}}$ we can evaluate the expectation value" + "With the OLS expressions for the optimal parameters $\\boldsymbol{\\hat{\\theta}}$ we can evaluate the expectation value" ] }, { "cell_type": "markdown", - "id": "186654ff", + "id": "ff486d1e", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E}(\\boldsymbol{\\hat{\\beta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\beta}=\\boldsymbol{\\beta}.\n", + "\\mathbb{E}(\\boldsymbol{\\hat{\\theta}}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\theta}=\\boldsymbol{\\theta}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "a37fa088", + "id": "e4307815", "metadata": { "editable": true }, @@ -286,35 +286,35 @@ "\n", "We can also calculate the variance\n", "\n", - "The variance of the optimal value $\\boldsymbol{\\hat{\\beta}}$ is" + "The variance of the optimal value $\\boldsymbol{\\hat{\\theta}}$ is" ] }, { "cell_type": "markdown", - "id": "11cb8014", + "id": "490b2cbf", "metadata": { "editable": true }, "source": [ "$$\n", "\\begin{eqnarray*}\n", - "\\mbox{Var}(\\boldsymbol{\\hat{\\beta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})] [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})]^{T} \\}\n", + "\\mbox{Var}(\\boldsymbol{\\hat{\\theta}}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})] [\\boldsymbol{\\theta} - \\mathbb{E}(\\boldsymbol{\\theta})]^{T} \\}\n", "\\\\\n", - "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}]^{T} \\}\n", + "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\theta}]^{T} \\}\n", "\\\\\n", - "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\\\\n", - "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "% \\\\\n", - "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", + "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", "% \\\\\n", - "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\boldsymbol{\\beta}^T\n", + "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\boldsymbol{\\theta}^T\n", "\\\\\n", - "& = & \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "& = & \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T}\n", "\\, \\, \\, = \\, \\, \\, \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1},\n", "\\end{eqnarray*}\n", "$$" @@ -322,21 +322,21 @@ }, { "cell_type": "markdown", - "id": "41169e60", + "id": "5d8dd6bc", "metadata": { "editable": true }, "source": [ "where we have used that $\\mathbb{E} (\\mathbf{Y} \\mathbf{Y}^{T}) =\n", - "\\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} +\n", - "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\beta}) = \\sigma^2\n", + "\\mathbf{X} \\, \\boldsymbol{\\theta} \\, \\boldsymbol{\\theta}^{T} \\, \\mathbf{X}^{T} +\n", + "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\theta}) = \\sigma^2\n", "\\, (\\mathbf{X}^{T} \\mathbf{X})^{-1}$, one obtains an estimate of the\n", "variance of the estimate of the $j$-th regression coefficient:\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $. This may be used to\n", "construct a confidence interval for the estimates.\n", "\n", "In a similar way, we can obtain analytical expressions for say the\n", - "expectation values of the parameters $\\boldsymbol{\\beta}$ and their variance\n", + "expectation values of the parameters $\\boldsymbol{\\theta}$ and their variance\n", "when we employ Ridge regression, allowing us again to define a confidence interval. \n", "\n", "It is rather straightforward to show that" @@ -344,80 +344,80 @@ }, { "cell_type": "markdown", - "id": "0883465c", + "id": "3594078a", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\beta}^{\\mathrm{OLS}}.\n", + "\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\theta}^{\\mathrm{OLS}}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "19c95f6a", + "id": "43007b7a", "metadata": { "editable": true }, "source": [ "We see clearly that \n", - "$\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\beta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", + "$\\mathbb{E} \\big[ \\boldsymbol{\\theta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\theta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", "\n", "We can also compute the variance as" ] }, { "cell_type": "markdown", - "id": "581281a4", + "id": "3cd4e7da", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", "$$" ] }, { "cell_type": "markdown", - "id": "0ba54fc9", + "id": "5816b21e", "metadata": { "editable": true }, "source": [ - "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\beta}$ goes to zero. \n", + "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\theta}$ goes to zero. \n", "\n", "With this, we can compute the difference" ] }, { "cell_type": "markdown", - "id": "7974083e", + "id": "b58b8607", "metadata": { "editable": true }, "source": [ "$$\n", - "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\beta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", + "\\mbox{Var}[\\boldsymbol{\\theta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\theta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "7ac0f724", + "id": "c843db7f", "metadata": { "editable": true }, "source": [ "The difference is non-negative definite since each component of the\n", "matrix product is non-negative definite. \n", - "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." + "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\theta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below." ] }, { "cell_type": "markdown", - "id": "b5ca9631", + "id": "e81b1c39", "metadata": { "editable": true }, @@ -431,28 +431,28 @@ "$\\sigma^2$.\n", "\n", "We found above that the outputs $\\boldsymbol{y}$ have a mean value given by\n", - "$\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$ and variance $\\sigma^2$. Since the entries to\n", + "$\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$ and variance $\\sigma^2$. Since the entries to\n", "the design matrix are not stochastic variables, we can assume that the\n", "probability distribution of our targets is also a normal distribution\n", - "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\beta}}$. This means that a\n", + "but now with mean value $\\boldsymbol{X}\\hat{\\boldsymbol{\\theta}}$. This means that a\n", "single output $y_i$ is given by the Gaussian distribution" ] }, { "cell_type": "markdown", - "id": "82ee5c7b", + "id": "ea9719e3", "metadata": { "editable": true }, "source": [ "$$\n", - "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "y_i\\sim \\mathcal{N}(\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta}, \\sigma^2)=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "1d6bf7a8", + "id": "c28b598f", "metadata": { "editable": true }, @@ -465,43 +465,43 @@ }, { "cell_type": "markdown", - "id": "7fb09ab6", + "id": "88063ed7", "metadata": { "editable": true }, "source": [ "$$\n", - "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]},\n", + "p(y_i, \\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]},\n", "$$" ] }, { "cell_type": "markdown", - "id": "26a08d2b", + "id": "a34e4c28", "metadata": { "editable": true }, "source": [ - "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\beta}$.\n", + "which reads as finding the likelihood of an event $y_i$ with the input variables $\\boldsymbol{X}$ given the parameters (to be determined) $\\boldsymbol{\\theta}$.\n", "\n", "Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event $\\boldsymbol{y}$ as the product of the single events, that is we have" ] }, { "cell_type": "markdown", - "id": "ef0ab4ee", + "id": "1fc7ae0a", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta}).\n", + "p(\\boldsymbol{y},\\boldsymbol{X}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}=\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta}).\n", "$$" ] }, { "cell_type": "markdown", - "id": "6ab207a9", + "id": "30d5a1be", "metadata": { "editable": true }, @@ -512,7 +512,7 @@ }, { "cell_type": "markdown", - "id": "dce1ba65", + "id": "1d29fd29", "metadata": { "editable": true }, @@ -524,7 +524,7 @@ }, { "cell_type": "markdown", - "id": "5ff3cdc2", + "id": "289a116c", "metadata": { "editable": true }, @@ -535,29 +535,29 @@ }, { "cell_type": "markdown", - "id": "9eea8701", + "id": "d1cfbb56", "metadata": { "editable": true }, "source": [ "$$\n", - "p(\\boldsymbol{D}\\vert\\boldsymbol{\\beta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\beta})^2}{2\\sigma^2}\\right]}.\n", + "p(\\boldsymbol{D}\\vert\\boldsymbol{\\theta})=\\prod_{i=0}^{n-1}\\frac{1}{\\sqrt{2\\pi\\sigma^2}}\\exp{\\left[-\\frac{(y_i-\\boldsymbol{X}_{i,*}\\boldsymbol{\\theta})^2}{2\\sigma^2}\\right]}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "ec4900c9", + "id": "e089fc0e", "metadata": { "editable": true }, "source": [ - "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\beta}$." + "It is a conditional probability (see below) and reads as the likelihood of a domain of events $\\boldsymbol{D}$ given a set of parameters $\\boldsymbol{\\theta}$." ] }, { "cell_type": "markdown", - "id": "723bf3b9", + "id": "32cf9944", "metadata": { "editable": true }, @@ -571,7 +571,7 @@ "data is the most probable. \n", "\n", "We will assume here that our events are given by the above Gaussian\n", - "distribution and we will determine the optimal parameters $\\beta$ by\n", + "distribution and we will determine the optimal parameters $\\theta$ by\n", "maximizing the above PDF. However, computing the derivatives of a\n", "product function is cumbersome and can easily lead to overflow and/or\n", "underflowproblems, with potentials for loss of numerical precision.\n", @@ -588,7 +588,7 @@ }, { "cell_type": "markdown", - "id": "f2794be9", + "id": "ee8a9544", "metadata": { "editable": true }, @@ -600,19 +600,19 @@ }, { "cell_type": "markdown", - "id": "11c65665", + "id": "4bfeb203", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\beta})},\n", + "C(\\boldsymbol{\\theta}=-\\log{\\prod_{i=0}^{n-1}p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})}=-\\sum_{i=0}^{n-1}\\log{p(y_i,\\boldsymbol{X}\\vert\\boldsymbol{\\theta})},\n", "$$" ] }, { "cell_type": "markdown", - "id": "8895b552", + "id": "256143f9", "metadata": { "editable": true }, @@ -622,63 +622,63 @@ }, { "cell_type": "markdown", - "id": "fe93f501", + "id": "60d75bb1", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{\\beta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})\\vert\\vert_2^2}{2\\sigma^2}.\n", + "C(\\boldsymbol{\\theta}=\\frac{n}{2}\\log{2\\pi\\sigma^2}+\\frac{\\vert\\vert (\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta})\\vert\\vert_2^2}{2\\sigma^2}.\n", "$$" ] }, { "cell_type": "markdown", - "id": "759dd1fb", + "id": "7b111014", "metadata": { "editable": true }, "source": [ - "Taking the derivative of the *new* cost function with respect to the parameters $\\beta$ we recognize our familiar OLS equation, namely" + "Taking the derivative of the *new* cost function with respect to the parameters $\\theta$ we recognize our familiar OLS equation, namely" ] }, { "cell_type": "markdown", - "id": "ef0a2ce5", + "id": "0f42a0b5", "metadata": { "editable": true }, "source": [ "$$\n", - "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right) =0,\n", + "\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\theta}\\right) =0,\n", "$$" ] }, { "cell_type": "markdown", - "id": "d9446627", + "id": "39a9276b", "metadata": { "editable": true }, "source": [ - "which leads to the well-known OLS equation for the optimal paramters $\\beta$" + "which leads to the well-known OLS equation for the optimal paramters $\\theta$" ] }, { "cell_type": "markdown", - "id": "b89fc4b8", + "id": "84c63927", "metadata": { "editable": true }, "source": [ "$$\n", - "\\hat{\\boldsymbol{\\beta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", + "\\hat{\\boldsymbol{\\theta}}^{\\mathrm{OLS}}=\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}!\n", "$$" ] }, { "cell_type": "markdown", - "id": "aaff973d", + "id": "c087ef22", "metadata": { "editable": true }, @@ -688,7 +688,7 @@ }, { "cell_type": "markdown", - "id": "5c0aef8a", + "id": "79503987", "metadata": { "editable": true }, @@ -707,7 +707,7 @@ }, { "cell_type": "markdown", - "id": "41eb0fd6", + "id": "c165025b", "metadata": { "editable": true }, @@ -735,7 +735,7 @@ }, { "cell_type": "markdown", - "id": "4c99db16", + "id": "efb63405", "metadata": { "editable": true }, @@ -761,7 +761,7 @@ }, { "cell_type": "markdown", - "id": "4cf4ddfb", + "id": "88d4bfd8", "metadata": { "editable": true }, @@ -778,7 +778,7 @@ }, { "cell_type": "markdown", - "id": "599f817a", + "id": "6a0795e0", "metadata": { "editable": true }, @@ -798,7 +798,7 @@ }, { "cell_type": "markdown", - "id": "faf3ddfc", + "id": "d815fdf3", "metadata": { "editable": true }, @@ -827,7 +827,7 @@ }, { "cell_type": "markdown", - "id": "1fa47e69", + "id": "a5b26ed1", "metadata": { "editable": true }, @@ -852,7 +852,7 @@ }, { "cell_type": "markdown", - "id": "02acff64", + "id": "e5783b81", "metadata": { "editable": true }, @@ -872,7 +872,7 @@ }, { "cell_type": "markdown", - "id": "2eb52e16", + "id": "9bdfff4f", "metadata": { "editable": true }, @@ -884,7 +884,7 @@ }, { "cell_type": "markdown", - "id": "1088a74a", + "id": "bd1e4a83", "metadata": { "editable": true }, @@ -894,7 +894,7 @@ }, { "cell_type": "markdown", - "id": "cc2591a0", + "id": "66d14a83", "metadata": { "editable": true }, @@ -909,7 +909,7 @@ }, { "cell_type": "markdown", - "id": "d2b75b23", + "id": "ac38d462", "metadata": { "editable": true }, @@ -922,7 +922,7 @@ }, { "cell_type": "markdown", - "id": "71339c96", + "id": "65488a60", "metadata": { "editable": true }, @@ -935,7 +935,7 @@ }, { "cell_type": "markdown", - "id": "b929fbd8", + "id": "9c8686d8", "metadata": { "editable": true }, @@ -947,7 +947,7 @@ }, { "cell_type": "markdown", - "id": "8623de93", + "id": "585dbaff", "metadata": { "editable": true }, @@ -960,7 +960,7 @@ }, { "cell_type": "markdown", - "id": "03d1d4b9", + "id": "a8093a3e", "metadata": { "editable": true }, @@ -971,7 +971,7 @@ }, { "cell_type": "markdown", - "id": "b886df76", + "id": "ec844baa", "metadata": { "editable": true }, @@ -985,7 +985,7 @@ }, { "cell_type": "markdown", - "id": "4865d692", + "id": "7f2b563b", "metadata": { "editable": true }, @@ -995,7 +995,7 @@ }, { "cell_type": "markdown", - "id": "58c1b3ef", + "id": "5a75bbe4", "metadata": { "editable": true }, @@ -1009,7 +1009,7 @@ }, { "cell_type": "markdown", - "id": "e81ade4c", + "id": "8d00a9ff", "metadata": { "editable": true }, @@ -1022,7 +1022,7 @@ }, { "cell_type": "markdown", - "id": "c2d28e00", + "id": "cae54a40", "metadata": { "editable": true }, @@ -1035,7 +1035,7 @@ }, { "cell_type": "markdown", - "id": "93051716", + "id": "ef87a15a", "metadata": { "editable": true }, @@ -1045,7 +1045,7 @@ }, { "cell_type": "markdown", - "id": "d2469547", + "id": "5bfaef5d", "metadata": { "editable": true }, @@ -1058,7 +1058,7 @@ }, { "cell_type": "markdown", - "id": "def0e1e3", + "id": "0aab5f59", "metadata": { "editable": true }, @@ -1068,7 +1068,7 @@ }, { "cell_type": "markdown", - "id": "553b0fd3", + "id": "71269afb", "metadata": { "editable": true }, @@ -1081,7 +1081,7 @@ }, { "cell_type": "markdown", - "id": "de16cecb", + "id": "4d809fab", "metadata": { "editable": true }, @@ -1093,7 +1093,7 @@ }, { "cell_type": "markdown", - "id": "5e7159b5", + "id": "734f7c5e", "metadata": { "editable": true }, @@ -1112,7 +1112,7 @@ }, { "cell_type": "markdown", - "id": "bb5e5e2e", + "id": "f4ff2861", "metadata": { "editable": true }, @@ -1125,7 +1125,7 @@ }, { "cell_type": "markdown", - "id": "b1004f8a", + "id": "fd66a47c", "metadata": { "editable": true }, @@ -1137,7 +1137,7 @@ }, { "cell_type": "markdown", - "id": "5b5548b4", + "id": "51e5434b", "metadata": { "editable": true }, @@ -1150,7 +1150,7 @@ }, { "cell_type": "markdown", - "id": "baec5167", + "id": "9129f7a5", "metadata": { "editable": true }, @@ -1170,7 +1170,7 @@ }, { "cell_type": "markdown", - "id": "e6902520", + "id": "e5fa7f9c", "metadata": { "editable": true }, @@ -1179,13 +1179,13 @@ "\n", "Confidence intervals are used in statistics and represent a type of estimate\n", "computed from the observed data. This gives a range of values for an\n", - "unknown parameter such as the parameters $\\boldsymbol{\\beta}$ from linear regression.\n", + "unknown parameter such as the parameters $\\boldsymbol{\\theta}$ from linear regression.\n", "\n", - "With the OLS expressions for the parameters $\\boldsymbol{\\beta}$ we found \n", - "$\\mathbb{E}(\\boldsymbol{\\beta}) = \\boldsymbol{\\beta}$, which means that the estimator of the regression parameters is unbiased.\n", + "With the OLS expressions for the parameters $\\boldsymbol{\\theta}$ we found \n", + "$\\mathbb{E}(\\boldsymbol{\\theta}) = \\boldsymbol{\\theta}$, which means that the estimator of the regression parameters is unbiased.\n", "\n", "In the exercises this week we show that the variance of the estimate of the $j$-th regression coefficient is\n", - "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\theta}_j ) = \\boldsymbol{\\sigma}^2 [(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} $.\n", "\n", "This quantity can be used to\n", "construct a confidence interval for the estimates." @@ -1193,34 +1193,34 @@ }, { "cell_type": "markdown", - "id": "33275440", + "id": "f5bbafe6", "metadata": { "editable": true }, "source": [ "## Standard Approach based on the Normal Distribution\n", "\n", - "We will assume that the parameters $\\beta$ follow a normal\n", + "We will assume that the parameters $\\theta$ follow a normal\n", "distribution. We can then define the confidence interval. Here we will be using as\n", - "shorthands $\\mu_{\\beta}$ for the above mean value and $\\sigma_{\\beta}$\n", + "shorthands $\\mu_{\\theta}$ for the above mean value and $\\sigma_{\\theta}$\n", "for the standard deviation. We have then a confidence interval" ] }, { "cell_type": "markdown", - "id": "778dc564", + "id": "0cc93a0c", "metadata": { "editable": true }, "source": [ "$$\n", - "\\left(\\mu_{\\beta}\\pm \\frac{z\\sigma_{\\beta}}{\\sqrt{n}}\\right),\n", + "\\left(\\mu_{\\theta}\\pm \\frac{z\\sigma_{\\theta}}{\\sqrt{n}}\\right),\n", "$$" ] }, { "cell_type": "markdown", - "id": "362371a5", + "id": "8dd4f5e0", "metadata": { "editable": true }, @@ -1240,18 +1240,18 @@ }, { "cell_type": "markdown", - "id": "ff19e8ba", + "id": "f51c546c", "metadata": { "editable": true }, "source": [ "## Resampling methods: Bootstrap background\n", "\n", - "Since $\\widehat{\\beta} = \\widehat{\\beta}(\\boldsymbol{X})$ is a function of random variables,\n", - "$\\widehat{\\beta}$ itself must be a random variable. Thus it has\n", + "Since $\\widehat{\\theta} = \\widehat{\\theta}(\\boldsymbol{X})$ is a function of random variables,\n", + "$\\widehat{\\theta}$ itself must be a random variable. Thus it has\n", "a pdf, call this function $p(\\boldsymbol{t})$. The aim of the bootstrap is to\n", "estimate $p(\\boldsymbol{t})$ by the relative frequency of\n", - "$\\widehat{\\beta}$. You can think of this as using a histogram\n", + "$\\widehat{\\theta}$. You can think of this as using a histogram\n", "in the place of $p(\\boldsymbol{t})$. If the relative frequency closely\n", "resembles $p(\\vec{t})$, then using numerics, it is straight forward to\n", "estimate all the interesting parameters of $p(\\boldsymbol{t})$ using point\n", @@ -1260,31 +1260,31 @@ }, { "cell_type": "markdown", - "id": "ed32c963", + "id": "e44fcf6d", "metadata": { "editable": true }, "source": [ "## Resampling methods: More Bootstrap background\n", "\n", - "In the case that $\\widehat{\\beta}$ has\n", + "In the case that $\\widehat{\\theta}$ has\n", "more than one component, and the components are independent, we use the\n", "same estimator on each component separately. If the probability\n", "density function of $X_i$, $p(x)$, had been known, then it would have\n", "been straightforward to do this by: \n", "1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \\cdots, X_n^*)$. \n", "\n", - "2. Then using these numbers, we could compute a replica of $\\widehat{\\beta}$ called $\\widehat{\\beta}^*$. \n", + "2. Then using these numbers, we could compute a replica of $\\widehat{\\theta}$ called $\\widehat{\\theta}^*$. \n", "\n", "By repeated use of the above two points, many\n", - "estimates of $\\widehat{\\beta}$ can be obtained. The\n", - "idea is to use the relative frequency of $\\widehat{\\beta}^*$\n", + "estimates of $\\widehat{\\theta}$ can be obtained. The\n", + "idea is to use the relative frequency of $\\widehat{\\theta}^*$\n", "(think of a histogram) as an estimate of $p(\\boldsymbol{t})$." ] }, { "cell_type": "markdown", - "id": "cc9dade9", + "id": "3bd69373", "metadata": { "editable": true }, @@ -1305,7 +1305,7 @@ }, { "cell_type": "markdown", - "id": "45ae3f5f", + "id": "e7f867d9", "metadata": { "editable": true }, @@ -1318,24 +1318,24 @@ "\n", "2. Define a vector $\\boldsymbol{x}^*$ containing the values which were drawn from $\\boldsymbol{x}$. \n", "\n", - "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\beta}^*$ by evaluating $\\widehat \\beta$ under the observations $\\boldsymbol{x}^*$. \n", + "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\theta}^*$ by evaluating $\\widehat \\theta$ under the observations $\\boldsymbol{x}^*$. \n", "\n", "4. Repeat this process $k$ times. \n", "\n", "When you are done, you can draw a histogram of the relative frequency\n", - "of $\\widehat \\beta^*$. This is your estimate of the probability\n", + "of $\\widehat \\theta^*$. This is your estimate of the probability\n", "distribution $p(t)$. Using this probability distribution you can\n", "estimate any statistics thereof. In principle you never draw the\n", - "histogram of the relative frequency of $\\widehat{\\beta}^*$. Instead\n", + "histogram of the relative frequency of $\\widehat{\\theta}^*$. Instead\n", "you use the estimators corresponding to the statistic of interest. For\n", "example, if you are interested in estimating the variance of $\\widehat\n", - "\\beta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", - "$\\widehat \\beta^*$." + "\\theta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", + "$\\widehat \\theta^*$." ] }, { "cell_type": "markdown", - "id": "0baf38d5", + "id": "5c2c3909", "metadata": { "editable": true }, @@ -1359,7 +1359,7 @@ { "cell_type": "code", "execution_count": 1, - "id": "f98e61ad", + "id": "a32faf6a", "metadata": { "collapsed": false, "editable": true @@ -1398,7 +1398,7 @@ }, { "cell_type": "markdown", - "id": "e96e18fe", + "id": "bc95505d", "metadata": { "editable": true }, @@ -1408,7 +1408,7 @@ }, { "cell_type": "markdown", - "id": "dddf1d57", + "id": "355f62af", "metadata": { "editable": true }, @@ -1419,7 +1419,7 @@ { "cell_type": "code", "execution_count": 2, - "id": "7c4757ce", + "id": "39fd1fa8", "metadata": { "collapsed": false, "editable": true @@ -1439,7 +1439,7 @@ }, { "cell_type": "markdown", - "id": "bcbeba22", + "id": "237b4db3", "metadata": { "editable": true }, @@ -1457,7 +1457,7 @@ }, { "cell_type": "markdown", - "id": "accffde7", + "id": "0c13e5b0", "metadata": { "editable": true }, @@ -1469,7 +1469,7 @@ }, { "cell_type": "markdown", - "id": "29b8983b", + "id": "a086aa7e", "metadata": { "editable": true }, @@ -1478,27 +1478,27 @@ "\n", "In our derivation of the ordinary least squares method we defined then\n", "an approximation to the function $f$ in terms of the parameters\n", - "$\\boldsymbol{\\beta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", - "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\beta}$. \n", + "$\\boldsymbol{\\theta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", + "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\theta}$. \n", "\n", - "Thereafter we found the parameters $\\boldsymbol{\\beta}$ by optimizing the means squared error via the so-called cost function" + "Thereafter we found the parameters $\\boldsymbol{\\theta}$ by optimizing the means squared error via the so-called cost function" ] }, { "cell_type": "markdown", - "id": "060299b9", + "id": "c3837d89", "metadata": { "editable": true }, "source": [ "$$\n", - "C(\\boldsymbol{X},\\boldsymbol{\\beta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", + "C(\\boldsymbol{X},\\boldsymbol{\\theta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", "$$" ] }, { "cell_type": "markdown", - "id": "4dea42f0", + "id": "7c7cd0a7", "metadata": { "editable": true }, @@ -1508,7 +1508,7 @@ }, { "cell_type": "markdown", - "id": "43d04a17", + "id": "b9db0cd5", "metadata": { "editable": true }, @@ -1520,7 +1520,7 @@ }, { "cell_type": "markdown", - "id": "0898fc00", + "id": "00482b2d", "metadata": { "editable": true }, @@ -1537,7 +1537,7 @@ }, { "cell_type": "markdown", - "id": "60c93378", + "id": "2e7c7291", "metadata": { "editable": true }, @@ -1549,7 +1549,7 @@ }, { "cell_type": "markdown", - "id": "50e2b540", + "id": "9a8d29b5", "metadata": { "editable": true }, @@ -1559,7 +1559,7 @@ }, { "cell_type": "markdown", - "id": "6286c56b", + "id": "9a8e2084", "metadata": { "editable": true }, @@ -1571,7 +1571,7 @@ }, { "cell_type": "markdown", - "id": "5141150d", + "id": "4a191c5c", "metadata": { "editable": true }, @@ -1581,7 +1581,7 @@ }, { "cell_type": "markdown", - "id": "ca57db29", + "id": "c37713f8", "metadata": { "editable": true }, @@ -1593,7 +1593,7 @@ }, { "cell_type": "markdown", - "id": "ed979b45", + "id": "82ca4a7b", "metadata": { "editable": true }, @@ -1603,7 +1603,7 @@ }, { "cell_type": "markdown", - "id": "11a712e8", + "id": "2765a841", "metadata": { "editable": true }, @@ -1619,7 +1619,7 @@ }, { "cell_type": "markdown", - "id": "b8415c0c", + "id": "321964c1", "metadata": { "editable": true }, @@ -1630,7 +1630,7 @@ { "cell_type": "code", "execution_count": 3, - "id": "581e2ad8", + "id": "5942226f", "metadata": { "collapsed": false, "editable": true @@ -1695,7 +1695,7 @@ }, { "cell_type": "markdown", - "id": "0ba21811", + "id": "cf2af19d", "metadata": { "editable": true }, @@ -1706,7 +1706,7 @@ { "cell_type": "code", "execution_count": 4, - "id": "41561f49", + "id": "7b371a0c", "metadata": { "collapsed": false, "editable": true @@ -1763,7 +1763,7 @@ }, { "cell_type": "markdown", - "id": "1c5f7f45", + "id": "6f583fcd", "metadata": { "editable": true }, @@ -1801,7 +1801,7 @@ }, { "cell_type": "markdown", - "id": "6a1dc288", + "id": "37766c51", "metadata": { "editable": true }, @@ -1828,7 +1828,7 @@ { "cell_type": "code", "execution_count": 5, - "id": "67251eaa", + "id": "0eac5ca9", "metadata": { "collapsed": false, "editable": true @@ -1890,7 +1890,7 @@ }, { "cell_type": "markdown", - "id": "3bcabf26", + "id": "d698096e", "metadata": { "editable": true }, @@ -1915,7 +1915,7 @@ }, { "cell_type": "markdown", - "id": "24248afc", + "id": "3e37a4d2", "metadata": { "editable": true }, @@ -1943,7 +1943,7 @@ }, { "cell_type": "markdown", - "id": "d089d4c1", + "id": "e7e43cf9", "metadata": { "editable": true }, @@ -1956,7 +1956,7 @@ { "cell_type": "code", "execution_count": 6, - "id": "4d0d1a95", + "id": "7aa6a568", "metadata": { "collapsed": false, "editable": true @@ -2056,7 +2056,7 @@ }, { "cell_type": "markdown", - "id": "1fc9687d", + "id": "9a90aec7", "metadata": { "editable": true }, @@ -2067,7 +2067,7 @@ { "cell_type": "code", "execution_count": 7, - "id": "8b5fdf71", + "id": "93b3af24", "metadata": { "collapsed": false, "editable": true @@ -2156,7 +2156,7 @@ }, { "cell_type": "markdown", - "id": "7013a0df", + "id": "90657c6d", "metadata": { "editable": true }, @@ -2166,7 +2166,7 @@ }, { "cell_type": "markdown", - "id": "96e0a896", + "id": "a88b86e7", "metadata": { "editable": true }, @@ -2179,7 +2179,7 @@ { "cell_type": "code", "execution_count": 8, - "id": "d0b17ea8", + "id": "89f45188", "metadata": { "collapsed": false, "editable": true @@ -2257,7 +2257,7 @@ }, { "cell_type": "markdown", - "id": "2b4e70e7", + "id": "75496517", "metadata": { "editable": true }, diff --git a/doc/src/week38/week38.do.txt b/doc/src/week38/week38.do.txt index f25419649..a95fd41be 100644 --- a/doc/src/week38/week38.do.txt +++ b/doc/src/week38/week38.do.txt @@ -37,7 +37,7 @@ move from a linear algebra analysis to a statistical analysis. In particular, we will focus on what the regularization terms can result in. We will amongst other things show that the regularization parameter can reduce considerably the variance of the parameters -$\beta$. +$\theta$. On of the advantages of doing linear regression is that we actually end up with @@ -60,7 +60,7 @@ independent, i.e.: The randomness of $\varepsilon_i$ implies that $\mathbf{y}_i$ is also a random variable. In particular, $\mathbf{y}_i$ is normally distributed, because $\varepsilon_i \sim -\mathcal{N}(0, \sigma^2)$ and $\mathbf{X}_{i,\ast} \, \bm{\beta}$ is a +\mathcal{N}(0, \sigma^2)$ and $\mathbf{X}_{i,\ast} \, \bm{\theta}$ is a non-random scalar. To specify the parameters of the distribution of $\mathbf{y}_i$ we need to calculate its first two moments. @@ -85,7 +85,7 @@ We approximate this function with our model from the solution of the linear regr function $f$ is approximated by $\bm{\tilde{y}}$ where we want to minimize $(\bm{y}-\bm{\tilde{y}})^2$, our MSE, with !bt \[ -\bm{\tilde{y}} = \bm{X}\bm{\beta}. +\bm{\tilde{y}} = \bm{X}\bm{\theta}. \] !et @@ -96,8 +96,8 @@ We can calculate the expectation value of $\bm{y}$ for a given element $i$ !bt \begin{align*} \mathbb{E}(y_i) & = -\mathbb{E}(\mathbf{X}_{i, \ast} \, \bm{\beta}) + \mathbb{E}(\varepsilon_i) -\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\mathbb{E}(\mathbf{X}_{i, \ast} \, \bm{\theta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \theta, \end{align*} !et while @@ -106,97 +106,97 @@ its variance is \begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i - \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - [\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, -\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \bm{\beta})^2 \\ & -= \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \bm{\beta})^2 + 2 \varepsilon_i -\mathbf{X}_{i, \ast} \, \bm{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, -\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \bm{\beta})^2 + 2 -\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \bm{\beta} + -\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \bm{\beta})^2 +\theta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \bm{\theta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \bm{\theta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \bm{\theta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \theta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \bm{\theta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \bm{\theta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \bm{\theta})^2 \\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, \mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. \end{align*} !et -Hence, $y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \bm{\beta}, \sigma^2)$, that is $\bm{y}$ follows a normal distribution with -mean value $\bm{X}\bm{\beta}$ and variance $\sigma^2$ (not be confused with the singular values of the SVD). +Hence, $y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \bm{\theta}, \sigma^2)$, that is $\bm{y}$ follows a normal distribution with +mean value $\bm{X}\bm{\theta}$ and variance $\sigma^2$ (not be confused with the singular values of the SVD). !split -===== Expectation value and variance for $\bm{\beta}$ ===== +===== Expectation value and variance for $\bm{\theta}$ ===== -With the OLS expressions for the optimal parameters $\bm{\hat{\beta}}$ we can evaluate the expectation value +With the OLS expressions for the optimal parameters $\bm{\hat{\theta}}$ we can evaluate the expectation value !bt \[ -\mathbb{E}(\bm{\hat{\beta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\bm{\beta}=\bm{\beta}. +\mathbb{E}(\bm{\hat{\theta}}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\bm{\theta}=\bm{\theta}. \] !et This means that the estimator of the regression parameters is unbiased. We can also calculate the variance -The variance of the optimal value $\bm{\hat{\beta}}$ is +The variance of the optimal value $\bm{\hat{\theta}}$ is !bt \begin{eqnarray*} -\mbox{Var}(\bm{\hat{\beta}}) & = & \mathbb{E} \{ [\bm{\beta} - \mathbb{E}(\bm{\beta})] [\bm{\beta} - \mathbb{E}(\bm{\beta})]^{T} \} +\mbox{Var}(\bm{\hat{\theta}}) & = & \mathbb{E} \{ [\bm{\theta} - \mathbb{E}(\bm{\theta})] [\bm{\theta} - \mathbb{E}(\bm{\theta})]^{T} \} \\ -& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \bm{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \bm{\beta}]^{T} \} +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \bm{\theta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \bm{\theta}]^{T} \} \\ -% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \bm{\beta} \, \bm{\beta}^{T} +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \bm{\theta} \, \bm{\theta}^{T} % \\ -% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \bm{\beta} \, \bm{\beta}^{T} +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \bm{\theta} \, \bm{\theta}^{T} % \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \bm{\beta} \, \bm{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \bm{\theta} \, \bm{\theta}^{T} \\ -& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \bm{\beta} \, \bm{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \bm{\beta} \, \bm{\beta}^{T} +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \bm{\theta} \, \bm{\theta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \bm{\theta} \, \bm{\theta}^{T} % \\ -% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \bm{\beta} \, \bm{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \bm{\theta} \, \bm{\theta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} % \\ -% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \bm{\beta} \bm{\beta}^T +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \bm{\theta} \bm{\theta}^T \\ -& = & \bm{\beta} \, \bm{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \bm{\beta} \, \bm{\beta}^{T} +& = & \bm{\theta} \, \bm{\theta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \bm{\theta} \, \bm{\theta}^{T} \, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, \end{eqnarray*} !et where we have used that $\mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = -\mathbf{X} \, \bm{\beta} \, \bm{\beta}^{T} \, \mathbf{X}^{T} + -\sigma^2 \, \mathbf{I}_{nn}$. From $\mbox{Var}(\bm{\beta}) = \sigma^2 +\mathbf{X} \, \bm{\theta} \, \bm{\theta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn}$. From $\mbox{Var}(\bm{\theta}) = \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}$, one obtains an estimate of the variance of the estimate of the $j$-th regression coefficient: -$\bm{\sigma}^2 (\bm{\beta}_j ) = \bm{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} $. This may be used to +$\bm{\sigma}^2 (\bm{\theta}_j ) = \bm{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} $. This may be used to construct a confidence interval for the estimates. In a similar way, we can obtain analytical expressions for say the -expectation values of the parameters $\bm{\beta}$ and their variance +expectation values of the parameters $\bm{\theta}$ and their variance when we employ Ridge regression, allowing us again to define a confidence interval. It is rather straightforward to show that !bt \[ -\mathbb{E} \big[ \bm{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\bm{\beta}^{\mathrm{OLS}}. +\mathbb{E} \big[ \bm{\theta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\bm{\theta}^{\mathrm{OLS}}. \] !et We see clearly that -$\mathbb{E} \big[ \bm{\beta}^{\mathrm{Ridge}} \big] \not= \bm{\beta}^{\mathrm{OLS}}$ for any $\lambda > 0$. We say then that the ridge estimator is biased. +$\mathbb{E} \big[ \bm{\theta}^{\mathrm{Ridge}} \big] \not= \bm{\theta}^{\mathrm{OLS}}$ for any $\lambda > 0$. We say then that the ridge estimator is biased. We can also compute the variance as !bt \[ -\mbox{Var}[\bm{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\mbox{Var}[\bm{\theta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, \] !et -and it is easy to see that if the parameter $\lambda$ goes to infinity then the variance of Ridge parameters $\bm{\beta}$ goes to zero. +and it is easy to see that if the parameter $\lambda$ goes to infinity then the variance of Ridge parameters $\bm{\theta}$ goes to zero. With this, we can compute the difference !bt \[ -\mbox{Var}[\bm{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\bm{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\mbox{Var}[\bm{\theta}^{\mathrm{OLS}}]-\mbox{Var}(\bm{\theta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. \] !et The difference is non-negative definite since each component of the matrix product is non-negative definite. -This means the variance we obtain with the standard OLS will always for $\lambda > 0$ be larger than the variance of $\bm{\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. +This means the variance we obtain with the standard OLS will always for $\lambda > 0$ be larger than the variance of $\bm{\theta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. !split @@ -209,15 +209,15 @@ distribution with zero mean value and an undetermined variance $\sigma^2$. We found above that the outputs $\bm{y}$ have a mean value given by -$\bm{X}\hat{\bm{\beta}}$ and variance $\sigma^2$. Since the entries to +$\bm{X}\hat{\bm{\theta}}$ and variance $\sigma^2$. Since the entries to the design matrix are not stochastic variables, we can assume that the probability distribution of our targets is also a normal distribution -but now with mean value $\bm{X}\hat{\bm{\beta}}$. This means that a +but now with mean value $\bm{X}\hat{\bm{\theta}}$. This means that a single output $y_i$ is given by the Gaussian distribution !bt \[ -y_i\sim \mathcal{N}(\bm{X}_{i,*}\bm{\beta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\beta})^2}{2\sigma^2}\right]}. +y_i\sim \mathcal{N}(\bm{X}_{i,*}\bm{\theta}, \sigma^2)=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\theta})^2}{2\sigma^2}\right]}. \] !et @@ -228,16 +228,16 @@ We assume now that the various $y_i$ values are stochastically distributed accor We define this distribution as !bt \[ -p(y_i, \bm{X}\vert\bm{\beta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\beta})^2}{2\sigma^2}\right]}, +p(y_i, \bm{X}\vert\bm{\theta})=\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\theta})^2}{2\sigma^2}\right]}, \] !et -which reads as finding the likelihood of an event $y_i$ with the input variables $\bm{X}$ given the parameters (to be determined) $\bm{\beta}$. +which reads as finding the likelihood of an event $y_i$ with the input variables $\bm{X}$ given the parameters (to be determined) $\bm{\theta}$. Since these events are assumed to be independent and identicall distributed we can build the probability distribution function (PDF) for all possible event $\bm{y}$ as the product of the single events, that is we have !bt \[ -p(\bm{y},\bm{X}\vert\bm{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\beta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\bm{X}\vert\bm{\beta}). +p(\bm{y},\bm{X}\vert\bm{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\theta})^2}{2\sigma^2}\right]}=\prod_{i=0}^{n-1}p(y_i,\bm{X}\vert\bm{\theta}). \] !et @@ -252,11 +252,11 @@ In the more general case the various inputs should be replaced by the possible f We can now rewrite the above probability as !bt \[ -p(\bm{D}\vert\bm{\beta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\beta})^2}{2\sigma^2}\right]}. +p(\bm{D}\vert\bm{\theta})=\prod_{i=0}^{n-1}\frac{1}{\sqrt{2\pi\sigma^2}}\exp{\left[-\frac{(y_i-\bm{X}_{i,*}\bm{\theta})^2}{2\sigma^2}\right]}. \] !et -It is a conditional probability (see below) and reads as the likelihood of a domain of events $\bm{D}$ given a set of parameters $\bm{\beta}$. +It is a conditional probability (see below) and reads as the likelihood of a domain of events $\bm{D}$ given a set of parameters $\bm{\theta}$. !split ===== Maximum Likelihood Estimation (MLE) ===== @@ -269,7 +269,7 @@ data is the most probable. We will assume here that our events are given by the above Gaussian -distribution and we will determine the optimal parameters $\beta$ by +distribution and we will determine the optimal parameters $\theta$ by maximizing the above PDF. However, computing the derivatives of a product function is cumbersome and can easily lead to overflow and/or underflowproblems, with potentials for loss of numerical precision. @@ -293,27 +293,27 @@ We could now define a new cost function to minimize, namely the negative logarit !bt \[ -C(\bm{\beta}=-\log{\prod_{i=0}^{n-1}p(y_i,\bm{X}\vert\bm{\beta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\bm{X}\vert\bm{\beta})}, +C(\bm{\theta}=-\log{\prod_{i=0}^{n-1}p(y_i,\bm{X}\vert\bm{\theta})}=-\sum_{i=0}^{n-1}\log{p(y_i,\bm{X}\vert\bm{\theta})}, \] !et which becomes !bt \[ -C(\bm{\beta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\bm{y}-\bm{X}\bm{\beta})\vert\vert_2^2}{2\sigma^2}. +C(\bm{\theta}=\frac{n}{2}\log{2\pi\sigma^2}+\frac{\vert\vert (\bm{y}-\bm{X}\bm{\theta})\vert\vert_2^2}{2\sigma^2}. \] !et -Taking the derivative of the *new* cost function with respect to the parameters $\beta$ we recognize our familiar OLS equation, namely +Taking the derivative of the *new* cost function with respect to the parameters $\theta$ we recognize our familiar OLS equation, namely !bt \[ -\bm{X}^T\left(\bm{y}-\bm{X}\bm{\beta}\right) =0, +\bm{X}^T\left(\bm{y}-\bm{X}\bm{\theta}\right) =0, \] !et -which leads to the well-known OLS equation for the optimal paramters $\beta$ +which leads to the well-known OLS equation for the optimal paramters $\theta$ !bt \[ -\hat{\bm{\beta}}^{\mathrm{OLS}}=\left(\bm{X}^T\bm{X}\right)^{-1}\bm{X}^T\bm{y}! +\hat{\bm{\theta}}^{\mathrm{OLS}}=\left(\bm{X}^T\bm{X}\right)^{-1}\bm{X}^T\bm{y}! \] !et @@ -601,13 +601,13 @@ $\tilde{p}(x)$. Confidence intervals are used in statistics and represent a type of estimate computed from the observed data. This gives a range of values for an -unknown parameter such as the parameters $\bm{\beta}$ from linear regression. +unknown parameter such as the parameters $\bm{\theta}$ from linear regression. -With the OLS expressions for the parameters $\bm{\beta}$ we found -$\mathbb{E}(\bm{\beta}) = \bm{\beta}$, which means that the estimator of the regression parameters is unbiased. +With the OLS expressions for the parameters $\bm{\theta}$ we found +$\mathbb{E}(\bm{\theta}) = \bm{\theta}$, which means that the estimator of the regression parameters is unbiased. In the exercises this week we show that the variance of the estimate of the $j$-th regression coefficient is -$\bm{\sigma}^2 (\bm{\beta}_j ) = \bm{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} $. +$\bm{\sigma}^2 (\bm{\theta}_j ) = \bm{\sigma}^2 [(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} $. This quantity can be used to construct a confidence interval for the estimates. @@ -616,14 +616,14 @@ construct a confidence interval for the estimates. !split ===== Standard Approach based on the Normal Distribution ===== -We will assume that the parameters $\beta$ follow a normal +We will assume that the parameters $\theta$ follow a normal distribution. We can then define the confidence interval. Here we will be using as -shorthands $\mu_{\beta}$ for the above mean value and $\sigma_{\beta}$ +shorthands $\mu_{\theta}$ for the above mean value and $\sigma_{\theta}$ for the standard deviation. We have then a confidence interval !bt \[ -\left(\mu_{\beta}\pm \frac{z\sigma_{\beta}}{\sqrt{n}}\right), +\left(\mu_{\theta}\pm \frac{z\sigma_{\theta}}{\sqrt{n}}\right), \] !et @@ -642,11 +642,11 @@ Bootstrap method, why it works and various theorems related to it. !split ===== Resampling methods: Bootstrap background ===== -Since $\widehat{\beta} = \widehat{\beta}(\bm{X})$ is a function of random variables, -$\widehat{\beta}$ itself must be a random variable. Thus it has +Since $\widehat{\theta} = \widehat{\theta}(\bm{X})$ is a function of random variables, +$\widehat{\theta}$ itself must be a random variable. Thus it has a pdf, call this function $p(\bm{t})$. The aim of the bootstrap is to estimate $p(\bm{t})$ by the relative frequency of -$\widehat{\beta}$. You can think of this as using a histogram +$\widehat{\theta}$. You can think of this as using a histogram in the place of $p(\bm{t})$. If the relative frequency closely resembles $p(\vec{t})$, then using numerics, it is straight forward to estimate all the interesting parameters of $p(\bm{t})$ using point @@ -656,17 +656,17 @@ estimators. !split ===== Resampling methods: More Bootstrap background ===== -In the case that $\widehat{\beta}$ has +In the case that $\widehat{\theta}$ has more than one component, and the components are independent, we use the same estimator on each component separately. If the probability density function of $X_i$, $p(x)$, had been known, then it would have been straightforward to do this by: o Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \cdots, X_n^*)$. -o Then using these numbers, we could compute a replica of $\widehat{\beta}$ called $\widehat{\beta}^*$. +o Then using these numbers, we could compute a replica of $\widehat{\theta}$ called $\widehat{\theta}^*$. By repeated use of the above two points, many -estimates of $\widehat{\beta}$ can be obtained. The -idea is to use the relative frequency of $\widehat{\beta}^*$ +estimates of $\widehat{\theta}$ can be obtained. The +idea is to use the relative frequency of $\widehat{\theta}^*$ (think of a histogram) as an estimate of $p(\bm{t})$. !split @@ -692,18 +692,18 @@ The independent bootstrap works like this: o Draw with replacement $n$ numbers for the observed variables $\bm{x} = (x_1,x_2,\cdots,x_n)$. o Define a vector $\bm{x}^*$ containing the values which were drawn from $\bm{x}$. -o Using the vector $\bm{x}^*$ compute $\widehat{\beta}^*$ by evaluating $\widehat \beta$ under the observations $\bm{x}^*$. +o Using the vector $\bm{x}^*$ compute $\widehat{\theta}^*$ by evaluating $\widehat \theta$ under the observations $\bm{x}^*$. o Repeat this process $k$ times. When you are done, you can draw a histogram of the relative frequency -of $\widehat \beta^*$. This is your estimate of the probability +of $\widehat \theta^*$. This is your estimate of the probability distribution $p(t)$. Using this probability distribution you can estimate any statistics thereof. In principle you never draw the -histogram of the relative frequency of $\widehat{\beta}^*$. Instead +histogram of the relative frequency of $\widehat{\theta}^*$. Instead you use the estimators corresponding to the statistic of interest. For example, if you are interested in estimating the variance of $\widehat -\beta$, apply the etsimator $\widehat \sigma^2$ to the values -$\widehat \beta^*$. +\theta$, apply the etsimator $\widehat \sigma^2$ to the values +$\widehat \theta^*$. !split @@ -791,13 +791,13 @@ where $\epsilon$ is normally distributed with mean zero and standard deviation $ In our derivation of the ordinary least squares method we defined then an approximation to the function $f$ in terms of the parameters -$\bm{\beta}$ and the design matrix $\bm{X}$ which embody our model, -that is $\bm{\tilde{y}}=\bm{X}\bm{\beta}$. +$\bm{\theta}$ and the design matrix $\bm{X}$ which embody our model, +that is $\bm{\tilde{y}}=\bm{X}\bm{\theta}$. -Thereafter we found the parameters $\bm{\beta}$ by optimizing the means squared error via the so-called cost function +Thereafter we found the parameters $\bm{\theta}$ by optimizing the means squared error via the so-called cost function !bt \[ -C(\bm{X},\bm{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\bm{y}-\bm{\tilde{y}})^2\right]. +C(\bm{X},\bm{\theta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\bm{y}-\bm{\tilde{y}})^2\right]. \] !et