-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathMaster_Thesis_Florian_Hochstrasser.aux
More file actions
263 lines (263 loc) · 31.5 KB
/
Copy pathMaster_Thesis_Florian_Hochstrasser.aux
File metadata and controls
263 lines (263 loc) · 31.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
\relax
\providecommand\hyper@newdestlabel[2]{}
\providecommand*\new@tpo@label[2]{}
\providecommand\HyperFirstAtBeginDocument{\AtBeginDocument}
\HyperFirstAtBeginDocument{\ifx\hyper@anchor\@undefined
\global\let\oldcontentsline\contentsline
\gdef\contentsline#1#2#3#4{\oldcontentsline{#1}{#2}{#3}}
\global\let\oldnewlabel\newlabel
\gdef\newlabel#1#2{\newlabelxx{#1}#2}
\gdef\newlabelxx#1#2#3#4#5#6{\oldnewlabel{#1}{{#2}{#3}}}
\AtEndDocument{\ifx\hyper@anchor\@undefined
\let\contentsline\oldcontentsline
\let\newlabel\oldnewlabel
\fi}
\fi}
\global\let\hyper@last\relax
\gdef\HyperFirstAtBeginDocument#1{#1}
\providecommand*\HyPL@Entry[1]{}
\HyPL@Entry{0<</S/D>>}
\babel@aux{english}{}
\@writefile{toc}{\contentsline {chapter}{\numberline {1}Introduction}{7}{chapter.1}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{intro}{{1}{7}{Introduction}{chapter.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {1.1}Task Background}{7}{section.1.1}\protected@file@percent }
\newlabel{task-background}{{1.1}{7}{Task Background}{section.1.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {1.2}Goals and Requirements}{8}{section.1.2}\protected@file@percent }
\newlabel{goals-and-requirements}{{1.2}{8}{Goals and Requirements}{section.1.2}{}}
\@writefile{toc}{\contentsline {section}{\numberline {1.3}Conventions and Notes}{8}{section.1.3}\protected@file@percent }
\newlabel{conventions-and-notes}{{1.3}{8}{Conventions and Notes}{section.1.3}{}}
\@writefile{toc}{\contentsline {chapter}{\numberline {2}Data}{9}{chapter.2}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{data}{{2}{9}{Data}{chapter.2}{}}
\@writefile{toc}{\contentsline {section}{\numberline {2.1}General Structure}{9}{section.2.1}\protected@file@percent }
\newlabel{general-structure}{{2.1}{9}{General Structure}{section.2.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {2.2}Exploratory Data Analysis}{9}{section.2.2}\protected@file@percent }
\newlabel{exploratory-data-analysis}{{2.2}{9}{Exploratory Data Analysis}{section.2.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {2.2.1}Data Types}{10}{subsection.2.2.1}\protected@file@percent }
\newlabel{data-types}{{2.2.1}{10}{Data Types}{subsection.2.2.1}{}}
\@writefile{lot}{\contentsline {table}{\numberline {\relax 2.1}{\ignorespaces Data types after import of raw csv data\relax }}{10}{table.caption.3}\protected@file@percent }
\providecommand*\caption@xref[2]{\@setref\relax\@undefined{#1}}
\newlabel{tab:data-desc}{{\relax 2.1}{10}{Data types after import of raw csv data\relax }{table.caption.3}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {2.2.2}Targets}{10}{subsection.2.2.2}\protected@file@percent }
\newlabel{targets}{{2.2.2}{10}{Targets}{subsection.2.2.2}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.1}{\ignorespaces Distribution of the binary target \(\text {TARGET}_B\).\relax }}{11}{figure.caption.4}\protected@file@percent }
\newlabel{fig:target-ratio}{{\relax 2.1}{11}{Distribution of the binary target \(\text {TARGET}_B\).\relax }{figure.caption.4}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.2}{\ignorespaces Distribution of \(\text {TARGET}_D\), the donation amount in \$ US (only amounts \textgreater {} 0.0 \$ are shown).\relax }}{11}{figure.caption.5}\protected@file@percent }
\newlabel{fig:target-d-distrib}{{\relax 2.2}{11}{Distribution of \(\text {TARGET}_D\), the donation amount in \$ US (only amounts \textgreater {} 0.0 \$ are shown).\relax }{figure.caption.5}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {2.2.3}Skewness}{12}{subsection.2.2.3}\protected@file@percent }
\newlabel{skewness}{{2.2.3}{12}{Skewness}{subsection.2.2.3}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.3}{\ignorespaces Fisher-Pearson standardized moment coefficient (G1) for all numeric features contained in the dataset. The confidence bound indicates the \(\mitalpha = 5 \%\) bound for the skewness of a normal distribution for any given feature.\relax }}{12}{figure.caption.6}\protected@file@percent }
\newlabel{fig:skew-all}{{\relax 2.3}{12}{Fisher-Pearson standardized moment coefficient (G1) for all numeric features contained in the dataset. The confidence bound indicates the \(\alpha = 5 \%\) bound for the skewness of a normal distribution for any given feature.\relax }{figure.caption.6}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.4}{\ignorespaces Least skewed features by G1 (adjusted Fisher-Pearson standardized moment coefficient).\relax }}{12}{figure.caption.7}\protected@file@percent }
\newlabel{fig:least-skewed}{{\relax 2.4}{12}{Least skewed features by G1 (adjusted Fisher-Pearson standardized moment coefficient).\relax }{figure.caption.7}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.5}{\ignorespaces Most skewed features by G1 (adjusted Fisher-Pearson standardized moment coefficient).\relax }}{13}{figure.caption.8}\protected@file@percent }
\newlabel{fig:most-skewed}{{\relax 2.5}{13}{Most skewed features by G1 (adjusted Fisher-Pearson standardized moment coefficient).\relax }{figure.caption.8}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {2.2.4}Correlations}{13}{subsection.2.2.4}\protected@file@percent }
\newlabel{correlations}{{2.2.4}{13}{Correlations}{subsection.2.2.4}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.6}{\ignorespaces Heatmap of feature correlations. Green means positive correlation, magenta means negative correlation. Perfect correlation occurs at 1.0 and -1.0.\relax }}{14}{figure.caption.9}\protected@file@percent }
\newlabel{fig:heatmap-all}{{\relax 2.6}{14}{Heatmap of feature correlations. Green means positive correlation, magenta means negative correlation. Perfect correlation occurs at 1.0 and -1.0.\relax }{figure.caption.9}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {2.2.5}Donation Patterns}{14}{subsection.2.2.5}\protected@file@percent }
\newlabel{donation-patterns}{{2.2.5}{14}{Donation Patterns}{subsection.2.2.5}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.7}{\ignorespaces Analysis of all-time RFA values by response to current promotion. \emph {Recency} is the time in months since the last donation, \emph {Frequency} the average number of donations per year and \emph {Amount} the average yearly donation amount.\relax }}{15}{figure.caption.10}\protected@file@percent }
\newlabel{fig:rfa-alltime}{{\relax 2.7}{15}{Analysis of all-time RFA values by response to current promotion. \emph {Recency} is the time in months since the last donation, \emph {Frequency} the average number of donations per year and \emph {Amount} the average yearly donation amount.\relax }{figure.caption.10}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.8}{\ignorespaces Donation amount for the current promotion against months since last donation. The dot size indicates the number of times an example has donated.\relax }}{15}{figure.caption.11}\protected@file@percent }
\newlabel{fig:donations-vs-time}{{\relax 2.8}{15}{Donation amount for the current promotion against months since last donation. The dot size indicates the number of times an example has donated.\relax }{figure.caption.11}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.9}{\ignorespaces Frequency of donations in the 13-24 months prior to current promotion against amount donated. Frequent donors give smaller amounts.\relax }}{16}{figure.caption.12}\protected@file@percent }
\newlabel{fig:rfa-f}{{\relax 2.9}{16}{Frequency of donations in the 13-24 months prior to current promotion against amount donated. Frequent donors give smaller amounts.\relax }{figure.caption.12}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.10}{\ignorespaces Geographical distribution of donations by zip code. Point size indicates total donations for a zip code while the hue shows average donation amount.\relax }}{17}{figure.caption.13}\protected@file@percent }
\newlabel{fig:donations-geo}{{\relax 2.10}{17}{Geographical distribution of donations by zip code. Point size indicates total donations for a zip code while the hue shows average donation amount.\relax }{figure.caption.13}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.11}{\ignorespaces Average cumulative donation amount per capita by living environment (C = city, U = urban, S = suburban, T = town, R = rural). The more rural, the higher the average donations.\relax }}{17}{figure.caption.14}\protected@file@percent }
\newlabel{fig:donations-le}{{\relax 2.11}{17}{Average cumulative donation amount per capita by living environment (C = city, U = urban, S = suburban, T = town, R = rural). The more rural, the higher the average donations.\relax }{figure.caption.14}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 2.12}{\ignorespaces Donation amount for current promotion by living environment and socio-economic status of examples. The violin plot shows the distribution of values similar to a kernel density estimation. Median values are indicated by white dots, the bold regions give the inner quartile range.\relax }}{18}{figure.caption.15}\protected@file@percent }
\newlabel{fig:donations-le-socioec}{{\relax 2.12}{18}{Donation amount for current promotion by living environment and socio-economic status of examples. The violin plot shows the distribution of values similar to a kernel density estimation. Median values are indicated by white dots, the bold regions give the inner quartile range.\relax }{figure.caption.15}{}}
\@writefile{toc}{\contentsline {chapter}{\numberline {3}Experimental Setup and Methods}{19}{chapter.3}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{experimental-setup-and-methods}{{3}{19}{Experimental Setup and Methods}{chapter.3}{}}
\@writefile{toc}{\contentsline {section}{\numberline {3.1}Tools Used}{19}{section.3.1}\protected@file@percent }
\newlabel{tools-used}{{3.1}{19}{Tools Used}{section.3.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {3.2}Data Handling}{19}{section.3.2}\protected@file@percent }
\newlabel{data-handling}{{3.2}{19}{Data Handling}{section.3.2}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 3.1}{\ignorespaces Data set use for training and predictions.\relax }}{20}{figure.caption.16}\protected@file@percent }
\newlabel{fig:data-splitting}{{\relax 3.1}{20}{Data set use for training and predictions.\relax }{figure.caption.16}{}}
\@writefile{toc}{\contentsline {section}{\numberline {3.3}Data Preprocessing}{20}{section.3.3}\protected@file@percent }
\newlabel{data-preprocessing}{{3.3}{20}{Data Preprocessing}{section.3.3}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.3.1}Cleaning}{20}{subsection.3.3.1}\protected@file@percent }
\newlabel{cleaning}{{3.3.1}{20}{Cleaning}{subsection.3.3.1}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.3.2}Feature Engineering}{21}{subsection.3.3.2}\protected@file@percent }
\newlabel{methods-feature-engineering}{{3.3.2}{21}{Feature Engineering}{subsection.3.3.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.3.3}Imputation}{21}{subsection.3.3.3}\protected@file@percent }
\newlabel{imputation}{{3.3.3}{21}{Imputation}{subsection.3.3.3}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.3.3.1}K-Nearest Neighbors}{22}{subsubsection.3.3.3.1}\protected@file@percent }
\newlabel{k-nearest-neighbors}{{3.3.3.1}{22}{K-Nearest Neighbors}{subsubsection.3.3.3.1}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.3.3.2}Iterative imputation}{22}{subsubsection.3.3.3.2}\protected@file@percent }
\newlabel{iterative-imputation}{{3.3.3.2}{22}{Iterative imputation}{subsubsection.3.3.3.2}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.3.3.3}Simple Imputation and Categorical Indicator}{22}{subsubsection.3.3.3.3}\protected@file@percent }
\newlabel{simple-imputation-and-categorical-indicator}{{3.3.3.3}{22}{Simple Imputation and Categorical Indicator}{subsubsection.3.3.3.3}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.3.4}Feature Selection}{23}{subsection.3.3.4}\protected@file@percent }
\newlabel{methods-feature-selection}{{3.3.4}{23}{Feature Selection}{subsection.3.3.4}{}}
\@writefile{toc}{\contentsline {section}{\numberline {3.4}Prediction}{23}{section.3.4}\protected@file@percent }
\newlabel{methods-prediction}{{3.4}{23}{Prediction}{section.3.4}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.4.1}Setup of the Two-Stage Prediction}{24}{subsection.3.4.1}\protected@file@percent }
\newlabel{setup-of-the-two-stage-prediction}{{3.4.1}{24}{Setup of the Two-Stage Prediction}{subsection.3.4.1}{}}
\newlabel{eq:y-b}{{\relax 3.1}{24}{Setup of the Two-Stage Prediction}{equation.3.4.1}{}}
\newlabel{eq:y-d}{{\relax 3.2}{24}{Setup of the Two-Stage Prediction}{equation.3.4.2}{}}
\newlabel{eq:indicator}{{\relax 3.3}{24}{Setup of the Two-Stage Prediction}{equation.3.4.3}{}}
\newlabel{eq:pi-alpha}{{\relax 3.4}{24}{Setup of the Two-Stage Prediction}{equation.3.4.4}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.4.2}Optimization of \(\mitalpha ^*\)}{25}{subsection.3.4.2}\protected@file@percent }
\newlabel{optimization-of-alpha}{{3.4.2}{25}{\texorpdfstring {Optimization of \(\alpha ^*\)}{Optimization of \textbackslash {}alpha\^{}*}}{subsection.3.4.2}{}}
\@writefile{toc}{\contentsline {section}{\numberline {3.5}Model Evaluation and -Selection}{25}{section.3.5}\protected@file@percent }
\newlabel{eval-and-select}{{3.5}{25}{Model Evaluation and -Selection}{section.3.5}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 3.2}{\ignorespaces Learning process schematic.\relax }}{25}{figure.caption.17}\protected@file@percent }
\newlabel{fig:evaluation-selection}{{\relax 3.2}{25}{Learning process schematic.\relax }{figure.caption.17}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.5.1}Evaluation}{26}{subsection.3.5.1}\protected@file@percent }
\newlabel{evaluation}{{3.5.1}{26}{Evaluation}{subsection.3.5.1}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.1.1}Randomized Grid Search}{26}{subsubsection.3.5.1.1}\protected@file@percent }
\newlabel{randomized-grid-search}{{3.5.1.1}{26}{Randomized Grid Search}{subsubsection.3.5.1.1}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.1.2}Cross-Validation}{26}{subsubsection.3.5.1.2}\protected@file@percent }
\newlabel{cross-validation}{{3.5.1.2}{26}{Cross-Validation}{subsubsection.3.5.1.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.5.2}Selection}{26}{subsection.3.5.2}\protected@file@percent }
\newlabel{selection}{{3.5.2}{26}{Selection}{subsection.3.5.2}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.2.1}Classifiers}{26}{subsubsection.3.5.2.1}\protected@file@percent }
\newlabel{classifiers}{{3.5.2.1}{26}{Classifiers}{subsubsection.3.5.2.1}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 3.3}{\ignorespaces Definition of the confusion matrix for a two-class problem.\relax }}{27}{figure.caption.18}\protected@file@percent }
\newlabel{fig:conf-mat-plot}{{\relax 3.3}{27}{Definition of the confusion matrix for a two-class problem.\relax }{figure.caption.18}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.2.2}Regressors}{28}{subsubsection.3.5.2.2}\protected@file@percent }
\newlabel{regressors}{{3.5.2.2}{28}{Regressors}{subsubsection.3.5.2.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.5.3}Dealing With Imbalanced Data}{28}{subsection.3.5.3}\protected@file@percent }
\newlabel{imblearn}{{3.5.3}{28}{Dealing With Imbalanced Data}{subsection.3.5.3}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.5.4}Algorithms}{28}{subsection.3.5.4}\protected@file@percent }
\newlabel{algorithms}{{3.5.4}{28}{Algorithms}{subsection.3.5.4}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.4.1}Random Forest}{28}{subsubsection.3.5.4.1}\protected@file@percent }
\newlabel{methods-rf}{{3.5.4.1}{28}{Random Forest}{subsubsection.3.5.4.1}{}}
\newlabel{eq:cart-const}{{\relax 3.5}{29}{Random Forest}{equation.3.5.5}{}}
\newlabel{eq:cart-hatc}{{\relax 3.6}{29}{Random Forest}{equation.3.5.6}{}}
\newlabel{eq:cart-opt}{{\relax 3.7}{29}{Random Forest}{equation.3.5.7}{}}
\newlabel{eq:cart-class}{{\relax 3.8}{29}{Random Forest}{equation.3.5.8}{}}
\newlabel{eq:cart-class-dec}{{\relax 3.9}{29}{Random Forest}{equation.3.5.9}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.4.2}Gradient Boosting Machine}{30}{subsubsection.3.5.4.2}\protected@file@percent }
\newlabel{methods-gbm}{{3.5.4.2}{30}{Gradient Boosting Machine}{subsubsection.3.5.4.2}{}}
\newlabel{eq:gbm-ensemble}{{\relax 3.10}{30}{Gradient Boosting Machine}{equation.3.5.10}{}}
\newlabel{eq:gbm-loss}{{\relax 3.11}{30}{Gradient Boosting Machine}{equation.3.5.11}{}}
\newlabel{eq:gbm-iterate}{{\relax 3.12}{31}{Gradient Boosting Machine}{equation.3.5.12}{}}
\newlabel{eq:gbm-grad}{{\relax 3.13}{31}{Gradient Boosting Machine}{equation.3.5.13}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.4.3}GLMnet}{31}{subsubsection.3.5.4.3}\protected@file@percent }
\newlabel{glmnet}{{3.5.4.3}{31}{GLMnet}{subsubsection.3.5.4.3}{}}
\newlabel{eq:glmnet-logit}{{\relax 3.14}{31}{GLMnet}{equation.3.5.14}{}}
\newlabel{eq:glmnet-gaussian}{{\relax 3.15}{32}{GLMnet}{equation.3.5.15}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.4.4}Multilayer Perceptron}{32}{subsubsection.3.5.4.4}\protected@file@percent }
\newlabel{multilayer-perceptron}{{3.5.4.4}{32}{Multilayer Perceptron}{subsubsection.3.5.4.4}{}}
\newlabel{eq:perceptron}{{\relax 3.16}{32}{Multilayer Perceptron}{equation.3.5.16}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 3.4}{\ignorespaces Neural network topology used. Two hidden layers \(\mathbf {h^{(1)}, h^{(2)}}\) are contained. \(\mathbf {b^{(2)}, b^{(3)}}\) and \(\mathbf {b^{(4)}}\) are the bias vectors for the respective layers.\relax }}{33}{figure.caption.19}\protected@file@percent }
\newlabel{fig:mlp-graph}{{\relax 3.4}{33}{Neural network topology used. Two hidden layers \(\mathbf {h^{(1)}, h^{(2)}}\) are contained. \(\mathbf {b^{(2)}, b^{(3)}}\) and \(\mathbf {b^{(4)}}\) are the bias vectors for the respective layers.\relax }{figure.caption.19}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.4.5}Support Vector Machine}{34}{subsubsection.3.5.4.5}\protected@file@percent }
\newlabel{support-vector-machine}{{3.5.4.5}{34}{Support Vector Machine}{subsubsection.3.5.4.5}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 3.5}{\ignorespaces Schematic display of an SVM hyperplane (in black), separating two overlapping classes. The margins are shown around the hyperplane, with support vectors falling on the margins. Misclassifications (examples on the wrong side of the hyperplane) have a total budget for distance from the separating plane. The margins are determined by respecting the budget. Adapted from Friedman, Hastie, and Tibshirani (\hyperlink {ref-friedman2001elements}{2001}).\relax }}{34}{figure.caption.20}\protected@file@percent }
\newlabel{fig:svm-schematic-plot}{{\relax 3.5}{34}{Schematic display of an SVM hyperplane (in black), separating two overlapping classes. The margins are shown around the hyperplane, with support vectors falling on the margins. Misclassifications (examples on the wrong side of the hyperplane) have a total budget for distance from the separating plane. The margins are determined by respecting the budget. Adapted from Friedman, Hastie, and Tibshirani (\protect \hyperlink {ref-friedman2001elements}{2001}).\relax }{figure.caption.20}{}}
\newlabel{eq:svm}{{\relax 3.18}{34}{Support Vector Machine}{equation.3.5.18}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {3.5.4.6}Bayesian Ridge Regression}{35}{subsubsection.3.5.4.6}\protected@file@percent }
\newlabel{bayesian-ridge-regression}{{3.5.4.6}{35}{Bayesian Ridge Regression}{subsubsection.3.5.4.6}{}}
\newlabel{eq:bayesion-linear-model-sk}{{\relax 3.19}{35}{Bayesian Ridge Regression}{equation.3.5.19}{}}
\newlabel{eq:bayesian-prob-model}{{\relax 3.20}{35}{Bayesian Ridge Regression}{equation.3.5.20}{}}
\newlabel{eq:prior-alpha}{{\relax 3.21}{35}{Bayesian Ridge Regression}{equation.3.5.21}{}}
\@writefile{toc}{\contentsline {chapter}{\numberline {4}Results and Discussion}{36}{chapter.4}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{results-and-discussion}{{4}{36}{Results and Discussion}{chapter.4}{}}
\@writefile{toc}{\contentsline {section}{\numberline {4.1}Preprocessing With Package kdd98}{36}{section.4.1}\protected@file@percent }
\newlabel{preprocessing-with-package-kdd98}{{4.1}{36}{Preprocessing With Package kdd98}{section.4.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {4.2}Imputation}{36}{section.4.2}\protected@file@percent }
\newlabel{imputation-1}{{4.2}{36}{Imputation}{section.4.2}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.1}{\ignorespaces Features with most (left) and fewest (right) missing values.\relax }}{37}{figure.caption.21}\protected@file@percent }
\newlabel{fig:most-fewest-missing}{{\relax 4.1}{37}{Features with most (left) and fewest (right) missing values.\relax }{figure.caption.21}{}}
\@writefile{toc}{\contentsline {section}{\numberline {4.3}Feature Selection}{37}{section.4.3}\protected@file@percent }
\newlabel{feature-selection}{{4.3}{37}{Feature Selection}{section.4.3}{}}
\@writefile{toc}{\contentsline {section}{\numberline {4.4}Model Evaluation and Selection}{38}{section.4.4}\protected@file@percent }
\newlabel{results-models}{{4.4}{38}{Model Evaluation and Selection}{section.4.4}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.4.1}Classifiers}{38}{subsection.4.4.1}\protected@file@percent }
\newlabel{classifiers-1}{{4.4.1}{38}{Classifiers}{subsection.4.4.1}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.2}{\ignorespaces Comparison of recall scores for all classifiers evaluated.\relax }}{39}{figure.caption.22}\protected@file@percent }
\newlabel{fig:recall-scores}{{\relax 4.2}{39}{Comparison of recall scores for all classifiers evaluated.\relax }{figure.caption.22}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.3}{\ignorespaces Comparison of ROC-AUC for the evaluated classifiers (left) and PR curves (right) for the classifiers.\relax }}{40}{figure.caption.23}\protected@file@percent }
\newlabel{fig:roc-auc-pr}{{\relax 4.3}{40}{Comparison of ROC-AUC for the evaluated classifiers (left) and PR curves (right) for the classifiers.\relax }{figure.caption.23}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.4}{\ignorespaces Confusion matrices for the 6 classifiers studied.\relax }}{41}{figure.caption.24}\protected@file@percent }
\newlabel{fig:conf-matrices}{{\relax 4.4}{41}{Confusion matrices for the 6 classifiers studied.\relax }{figure.caption.24}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.5}{\ignorespaces Coefficient values for GLMnet.\relax }}{41}{figure.caption.25}\protected@file@percent }
\newlabel{fig:glmnet-coefficients}{{\relax 4.5}{41}{Coefficient values for GLMnet.\relax }{figure.caption.25}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.6}{\ignorespaces Feature importances determined with the RF classifier. Impurity is measured by Gini importance / mean decrease impurity. Error bars give bootstrap error on 50 repetitions.\relax }}{42}{figure.caption.26}\protected@file@percent }
\newlabel{fig:importances}{{\relax 4.6}{42}{Feature importances determined with the RF classifier. Impurity is measured by Gini importance / mean decrease impurity. Error bars give bootstrap error on 50 repetitions.\relax }{figure.caption.26}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.4.2}Regressors}{42}{subsection.4.4.2}\protected@file@percent }
\newlabel{regressors-1}{{4.4.2}{42}{Regressors}{subsection.4.4.2}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.7}{\ignorespaces Target before transformation (left) and after a Box-Cox transformation (right).\relax }}{42}{figure.caption.27}\protected@file@percent }
\newlabel{fig:reg-targ-transform}{{\relax 4.7}{42}{Target before transformation (left) and after a Box-Cox transformation (right).\relax }{figure.caption.27}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.8}{\ignorespaces Distribution of (Box-Cox transformed) donation amounts for the four regressors evaluated.\relax }}{43}{figure.caption.28}\protected@file@percent }
\newlabel{fig:reg-distrib}{{\relax 4.8}{43}{Distribution of (Box-Cox transformed) donation amounts for the four regressors evaluated.\relax }{figure.caption.28}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.9}{\ignorespaces Evaluation metric \(R^2\) for all regression models evaluated. The domain for \(R^2\) is \((-\qopname \relax m{inf}, 1]\).)\relax }}{43}{figure.caption.29}\protected@file@percent }
\newlabel{fig:reg-eval}{{\relax 4.9}{43}{Evaluation metric \(R^2\) for all regression models evaluated. The domain for \(R^2\) is \((-\inf , 1]\).)\relax }{figure.caption.29}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.10}{\ignorespaces Feature importances for the RF regressor.\relax }}{44}{figure.caption.30}\protected@file@percent }
\newlabel{fig:reg-importance}{{\relax 4.10}{44}{Feature importances for the RF regressor.\relax }{figure.caption.30}{}}
\@writefile{toc}{\contentsline {section}{\numberline {4.5}Prediction}{44}{section.4.5}\protected@file@percent }
\newlabel{prediction}{{4.5}{44}{Prediction}{section.4.5}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.5.1}Prediction of Donation Probability}{44}{subsection.4.5.1}\protected@file@percent }
\newlabel{prediction-of-donation-probability}{{4.5.1}{44}{Prediction of Donation Probability}{subsection.4.5.1}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.11}{\ignorespaces Distribution of predicted donation probabilities \(\hat {y}_b\) on the learning data set.\relax }}{44}{figure.caption.31}\protected@file@percent }
\newlabel{fig:y-b-predict}{{\relax 4.11}{44}{Distribution of predicted donation probabilities \(\hat {y}_b\) on the learning data set.\relax }{figure.caption.31}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.5.2}Conditional Prediction of the Donation Amount}{45}{subsection.4.5.2}\protected@file@percent }
\newlabel{conditional-prediction-of-the-donation-amount}{{4.5.2}{45}{Conditional Prediction of the Donation Amount}{subsection.4.5.2}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.12}{\ignorespaces Conditionally predicted donation amounts, Box-Cox transformed (left) and on the original scale (right).\relax }}{45}{figure.caption.32}\protected@file@percent }
\newlabel{fig:y-d-predict}{{\relax 4.12}{45}{Conditionally predicted donation amounts, Box-Cox transformed (left) and on the original scale (right).\relax }{figure.caption.32}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.5.3}Profit Optimization}{45}{subsection.4.5.3}\protected@file@percent }
\newlabel{profit-optimization}{{4.5.3}{45}{Profit Optimization}{subsection.4.5.3}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {\relax 4.13}{\ignorespaces Expected profit for a range of \(\mitalpha \) values in \([0,1]\) with overlayed cubic spline and polynomial function of order 12.\relax }}{46}{figure.caption.33}\protected@file@percent }
\newlabel{fig:alpha-grid}{{\relax 4.13}{46}{Expected profit for a range of \(\alpha \) values in \([0,1]\) with overlayed cubic spline and polynomial function of order 12.\relax }{figure.caption.33}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.5.4}Final Prediction}{46}{subsection.4.5.4}\protected@file@percent }
\newlabel{final-prediction}{{4.5.4}{46}{Final Prediction}{subsection.4.5.4}{}}
\@writefile{lot}{\contentsline {table}{\numberline {\relax 4.1}{\ignorespaces Prediction results for the test data set (in color) and the results of the cup-winners. \(N^*\) denotes number of examples selected. The theoretical maximum was calculated with: \$ (\(\sum _{i=1}^n \symbb {1}_{\{\text {TARGET}_{D,i} > 0.0\}}*(\text {TARGET}_{D,i} - u)\) with \(u\) the unit cost of 0.68 \$ per mailing).\relax }}{47}{table.caption.34}\protected@file@percent }
\newlabel{tab:prediction-results}{{\relax 4.1}{47}{Prediction results for the test data set (in color) and the results of the cup-winners. \(N^*\) denotes number of examples selected. The theoretical maximum was calculated with: \$ (\(\sum _{i=1}^n \mathbb {1}_{\{\text {TARGET}_{D,i} > 0.0\}}*(\text {TARGET}_{D,i} - u)\) with \(u\) the unit cost of 0.68 \$ per mailing).\relax }{table.caption.34}{}}
\@writefile{toc}{\contentsline {chapter}{\numberline {5}Conclusions}{48}{chapter.5}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{conclusions}{{5}{48}{Conclusions}{chapter.5}{}}
\newlabel{references}{{5}{49}{References}{chapter*.35}{}}
\@writefile{toc}{\contentsline {chapter}{References}{49}{chapter*.35}\protected@file@percent }
\@writefile{toc}{\contentsline {chapter}{\numberline {A}Software}{52}{appendix.A}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{software}{{A}{52}{Software}{appendix.A}{}}
\@writefile{toc}{\contentsline {section}{\numberline {A.1}Python Environment}{52}{section.A.1}\protected@file@percent }
\newlabel{python-environment}{{A.1}{52}{Python Environment}{section.A.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {A.2}Package kdd98}{52}{section.A.2}\protected@file@percent }
\newlabel{package-kdd98}{{A.2}{52}{Package kdd98}{section.A.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {A.2.1}Usage}{52}{subsection.A.2.1}\protected@file@percent }
\newlabel{usage}{{A.2.1}{52}{Usage}{subsection.A.2.1}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {A.2.1.1}Data Provisioning}{52}{subsubsection.A.2.1.1}\protected@file@percent }
\newlabel{data-provisioning}{{A.2.1.1}{52}{Data Provisioning}{subsubsection.A.2.1.1}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {A.2.1.2}Predictions}{53}{subsubsection.A.2.1.2}\protected@file@percent }
\newlabel{predictions}{{A.2.1.2}{53}{Predictions}{subsubsection.A.2.1.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {A.2.2}Installation}{53}{subsection.A.2.2}\protected@file@percent }
\newlabel{installation}{{A.2.2}{53}{Installation}{subsection.A.2.2}{}}
\@writefile{toc}{\contentsline {subsubsection}{\numberline {A.2.2.1}Project structure}{53}{subsubsection.A.2.2.1}\protected@file@percent }
\newlabel{project-structure}{{A.2.2.1}{53}{Project structure}{subsubsection.A.2.2.1}{}}
\@writefile{toc}{\contentsline {chapter}{\numberline {B}KDD Cup Documents}{54}{appendix.B}\protected@file@percent }
\@writefile{lof}{\addvspace {10\p@ }}
\@writefile{lot}{\addvspace {10\p@ }}
\@writefile{lol}{\addvspace {10\p@ }}
\newlabel{kdd-cup-documents}{{B}{54}{KDD Cup Documents}{appendix.B}{}}
\@writefile{toc}{\contentsline {section}{\numberline {B.1}Cup Documentation}{54}{section.B.1}\protected@file@percent }
\newlabel{data-set-documentation}{{B.1}{54}{Cup Documentation}{section.B.1}{}}
\@writefile{toc}{\contentsline {section}{\numberline {B.2}Data Set Dictionary}{79}{section.B.2}\protected@file@percent }
\newlabel{data-set-dictionary}{{B.2}{79}{Data Set Dictionary}{section.B.2}{}}
\global\csname @altsecnumformattrue\endcsname
\global\@namedef{scr@dte@chapter@lastmaxnumwidth}{12.7163pt}
\global\@namedef{scr@dte@section@lastmaxnumwidth}{23.23767pt}
\global\@namedef{scr@dte@subsection@lastmaxnumwidth}{31.45016pt}
\global\@namedef{scr@dte@figure@lastmaxnumwidth}{26.27992pt}