{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.4.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30749,"isInternetEnabled":true,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n# Code om figuren aan te passen in output\nfig <- function(width, heigth){\n    # borrowed from https://www.kaggle.com/getting-started/105201\n    options(repr.plot.width = width, repr.plot.height = heigth)\n}\n\nsuppressPackageStartupMessages({\nlibrary(tidyverse, quietly = TRUE)\nlibrary(janitor)\nlibrary (caret)\nlibrary (randomForest)\nlibrary (naniar)\nlibrary(gridExtra)\nlibrary(DT)\nlibrary (htmltools)\nlibrary(recipes)\nlibrary (tidymodels)\nlibrary (corrplot)\nlibrary(tidymodels)\nlibrary(reticulate)\nlibrary(lightgbm)\nlibrary(bonsai)\nlibrary(finetune)\nlibrary(e1071) \nlibrary(MASS)\nlibrary(pheatmap)\nlibrary(h2o)\nlibrary(agua)\n})","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T19:47:28.192106Z","iopub.execute_input":"2024-12-08T19:47:28.194984Z","iopub.status.idle":"2024-12-08T19:47:35.413894Z","shell.execute_reply":"2024-12-08T19:47:35.411574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Notebook version\n1. EDA + baseline in H2O. Sample (n=100.000)\n2. Let's try a simple solution just ordinary Penalized regression based on results other notebooks\n3. Let's try GLm with H2O\n4. Add 'ouderdom' to set","metadata":{}},{"cell_type":"markdown","source":"# Import data","metadata":{}},{"cell_type":"code","source":"df.train=read.csv('/kaggle/input/playground-series-s4e12/train.csv')%>%clean_names()\ndf.test=read.csv('/kaggle/input/playground-series-s4e12/test.csv')%>%clean_names()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T19:49:04.145772Z","iopub.execute_input":"2024-12-08T19:49:04.178832Z","iopub.status.idle":"2024-12-08T19:49:43.307995Z","shell.execute_reply":"2024-12-08T19:49:43.305873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Structure\ndim(df.train)\nstr (df.train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:48:27.350515Z","iopub.execute_input":"2024-12-08T13:48:27.352094Z","iopub.status.idle":"2024-12-08T13:48:27.390664Z","shell.execute_reply":"2024-12-08T13:48:27.388424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Tidy data","metadata":{}},{"cell_type":"code","source":"# Calculate unique values\nfor (i in colnames (df.train)){ \nprint (paste0(i, \"=\", class(df.train[, i]), \", unique values are:\", (n_distinct (df.train[, i]))))}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:26:54.627825Z","iopub.execute_input":"2024-12-07T08:26:54.629424Z","iopub.status.idle":"2024-12-07T08:26:55.443207Z","shell.execute_reply":"2024-12-07T08:26:55.440866Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's take the year from policy_start_date","metadata":{}},{"cell_type":"code","source":"df.train$policy_start_date=ymd_hms(df.train$policy_start_date)\ndf.test$policy_start_date= ymd_hms(df.test$policy_start_date)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:48:42.937174Z","iopub.execute_input":"2024-12-08T13:48:42.938758Z","iopub.status.idle":"2024-12-08T13:48:43.887899Z","shell.execute_reply":"2024-12-08T13:48:43.886101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Let's take a peak**","metadata":{}},{"cell_type":"code","source":"# Let's create a list with all the possible values\nlong_df_train <- df.train %>%dplyr::select(where(is.character))%>% \n  pivot_longer(\n    cols = `gender`:`property_type`, \n    names_to = \"col\",\n    values_to = \"values\"\n)\nextreem_train=long_df_train%>%group_by (col, values)%>%summarise(Aantal=n())\nextreem_train=extreem_train%>%mutate (Percent=round(Aantal/nrow(df.train)*100,digits=2))%>%arrange (col,desc(Aantal))\ndatatable(extreem_train, filter = 'top', options = list(\n  pageLength = 5, autoWidth = TRUE,editable = 'cell'\n))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:28:14.4959Z","iopub.execute_input":"2024-12-07T08:28:14.497489Z","iopub.status.idle":"2024-12-07T08:28:15.902592Z","shell.execute_reply":"2024-12-07T08:28:15.899893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"long_df_test <- df.test %>%dplyr::select(where(is.character))%>% \n  pivot_longer(\n    cols = `gender`:`property_type`, \n    names_to = \"col\",\n    values_to = \"values\"\n)\nextreem_test=long_df_test%>%group_by (col, values)%>%summarise(Aantal=n())\nextreem_test=extreem_test%>%mutate (Percent=round(Aantal/nrow(df.train)*100,digits=2))%>%arrange (col,desc(Aantal))\ndatatable(extreem_test, filter = 'top', options = list(\n  pageLength = 5, autoWidth = TRUE,editable = 'cell'\n))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:28:24.083693Z","iopub.execute_input":"2024-12-07T08:28:24.085425Z","iopub.status.idle":"2024-12-07T08:28:24.951363Z","shell.execute_reply":"2024-12-07T08:28:24.948685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# values only present in df.train\nextreem_train%>%anti_join(extreem_test, by=join_by(\"col\", \"values\"))%>%dplyr::select ('col', 'values')%>%count()\n# values only present in df.test\nextreem_test%>%anti_join(extreem_train, by=join_by(\"col\", \"values\"))%>%dplyr::select ('col', 'values')%>%count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:28:33.863803Z","iopub.execute_input":"2024-12-07T08:28:33.865398Z","iopub.status.idle":"2024-12-07T08:28:33.922514Z","shell.execute_reply":"2024-12-07T08:28:33.920163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Conclusion: Besides policy_start_date all the values from training are in the file test a vice vers,","metadata":{}},{"cell_type":"markdown","source":"**Unique values**\nLet's find out if their are duplicate records in the file","metadata":{}},{"cell_type":"code","source":"df.train[2:20] %>%\n  group_by_all() %>%\n  filter(n() > 1) %>%\n  ungroup()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:28:42.306045Z","iopub.execute_input":"2024-12-07T08:28:42.307838Z","iopub.status.idle":"2024-12-07T08:28:56.738753Z","shell.execute_reply":"2024-12-07T08:28:56.736364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Missings**","metadata":{}},{"cell_type":"code","source":"# Which features have missings\ndf.train %>% miss_var_which()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:29:01.101374Z","iopub.execute_input":"2024-12-07T08:29:01.103055Z","iopub.status.idle":"2024-12-07T08:29:01.237566Z","shell.execute_reply":"2024-12-07T08:29:01.235094Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So most off the numeric features have missings ","metadata":{}},{"cell_type":"code","source":"fig (12,12)\nna_variables <- df.train %>% miss_var_which()\nna_upset_plot_train <- gg_miss_upset(\n  df.train %>% dplyr::select(all_of(na_variables)),\n  main.bar.color = \"#55c8a6\",\n  sets.bar.color = \"#55c8a6\",\n  matrix.color = \"#55c8a6\",\n  shade.color = \"#55c8a6\",\n  nsets = length(na_variables),\n  order.by = \"freq\",\n  point.size = 2,\n  line.size = 0.75,\n  shade.alpha = 0.2,\n  text.scale = c(1, 1, 0.8, 1, 0.8, 0.8)\n)\nna_upset_plot_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:29:11.152347Z","iopub.execute_input":"2024-12-07T08:29:11.154007Z","iopub.status.idle":"2024-12-07T08:29:13.580159Z","shell.execute_reply":"2024-12-07T08:29:13.578199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fig (8,8)\n#n_miss (df.train)\n#naniar::mcar_test (df.train) # Little's MCAR test\n#gg_miss_var (df.train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T07:18:57.531925Z","iopub.execute_input":"2024-12-07T07:18:57.533304Z","iopub.status.idle":"2024-12-07T07:18:57.543933Z","shell.execute_reply":"2024-12-07T07:18:57.542247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So the missings have no pattern ","metadata":{}},{"cell_type":"markdown","source":"# Split data","metadata":{}},{"cell_type":"code","source":"# Splitsen data\n\nset.seed(50) # setting the seed for the sample function\ndf.train.split <- sample(2\n                        , nrow(df.train)\n                        , replace = TRUE\n                        , prob = c(0.85, 0.15))\ndf_train = df.train[df.train.split == 1,]\ndf_val   = df.train[df.train.split == 2,]\n\npaste('Lengte trainingset: ',nrow(df_train),'| Lengte validatieset : ',nrow(df_val),'| Length Test Set: ',nrow(df.test))\nhead(df_train )\nhead (df_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:48:58.598925Z","iopub.execute_input":"2024-12-08T13:48:58.600587Z","iopub.status.idle":"2024-12-08T13:48:59.826119Z","shell.execute_reply":"2024-12-08T13:48:59.824442Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"**Target**","metadata":{}},{"cell_type":"code","source":"\n#De moments of Premium amount are:\nsummary(df_train$premium_amount)\npaste('Skewness of premium_amount is: ',skewness(df_train$premium_amount))\nmyhist <- hist(df_train$premium_amount)\nmultiplier <- myhist$counts / myhist$density\nmydensity <- density(df_train$premium_amount)\nmydensity$y <- mydensity$y * multiplier[1]\n\nplot(myhist)\nlines(mydensity)\nmyx <- seq(min(df_train$premium_amount), max(df_train$premium_amount), length.out= 100)\nmymean <- mean(df_train$premium_amount)\nmysd <- sd(df_train$premium_amount)\n\nnormal <- dnorm(x = myx, mean = mymean, sd = mysd)\nlines(myx, normal * multiplier[1], col = \"blue\", lwd = 2)\n\nsd_x <- seq(mymean - 3 * mysd, mymean + 3 * mysd, by = mysd)\nsd_y <- dnorm(x = sd_x, mean = mymean, sd = mysd) * multiplier[1]\n\nsegments(x0 = sd_x, y0= 0, x1 = sd_x, y1 = sd_y, col = \"firebrick4\", lwd = 2)\n\nqqnorm( (df_train$premium_amount), pch = 1, frame = FALSE)\nqqline( (df_train$premium_amount), col = \"steelblue\", lwd = 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:29:32.94704Z","iopub.execute_input":"2024-12-07T08:29:32.948666Z","iopub.status.idle":"2024-12-07T08:30:23.122673Z","shell.execute_reply":"2024-12-07T08:30:23.119339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nlibrary(MASS)\nx=df_train$premium_amount\nb=boxcox(lm(x ~ 1))\nlambda <- b$x[which.max(b$y)]\ncat ('The Lamda of this transformation =', lambda)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:30:35.207264Z","iopub.execute_input":"2024-12-07T08:30:35.208863Z","iopub.status.idle":"2024-12-07T08:30:40.424111Z","shell.execute_reply":"2024-12-07T08:30:40.42226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library (goft)\ngamma_test(df.train$premium_amount)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:30:50.404967Z","iopub.execute_input":"2024-12-07T08:30:50.406527Z","iopub.status.idle":"2024-12-07T08:30:50.721409Z","shell.execute_reply":"2024-12-07T08:30:50.719508Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**What happens with log transformation**","metadata":{}},{"cell_type":"code","source":"#De moments of Premium amount are:\nsummary(log (df_train$premium_amount))\npaste('Skewness of premium_amount is: ',skewness(log(df_train$premium_amount)))\nmyhist <- hist(log (df_train$premium_amount))\nmultiplier <- myhist$counts / myhist$density\nmydensity <- density(log (df_train$premium_amount))\nmydensity$y <- mydensity$y * multiplier[1]\n\nplot(myhist)\nlines(mydensity)\nmyx <- seq(min(log (df_train$premium_amount)), max(log (df_train$premium_amount)), length.out= 100)\nmymean <- mean(log (df_train$premium_amount))\nmysd <- sd(log (df_train$premium_amount))\n\nnormal <- dnorm(x = myx, mean = mymean, sd = mysd)\nlines(myx, normal * multiplier[1], col = \"blue\", lwd = 2)\n\nsd_x <- seq(mymean - 3 * mysd, mymean + 3 * mysd, by = mysd)\nsd_y <- dnorm(x = sd_x, mean = mymean, sd = mysd) * multiplier[1]\n\nsegments(x0 = sd_x, y0= 0, x1 = sd_x, y1 = sd_y, col = \"firebrick4\", lwd = 2)\n\nqqnorm( (log(df_train$premium_amount)), pch = 1, frame = FALSE)\nqqline( (log(df_train$premium_amount)), col = \"steelblue\", lwd = 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:31:00.641649Z","iopub.execute_input":"2024-12-07T08:31:00.643314Z","iopub.status.idle":"2024-12-07T08:31:49.961218Z","shell.execute_reply":"2024-12-07T08:31:49.959286Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's change the target to get better cost function","metadata":{}},{"cell_type":"code","source":"df.train$premium_amount=log (df.train$premium_amount)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Some insides numuric features**","metadata":{}},{"cell_type":"code","source":"all=subset( df_train, select = -c(id, premium_amount) )\nnumericVars <- which(sapply(all, is.numeric)) #index vector numeric variables\nnumericVarNames <- names(numericVars) #saving names vector for use later on\ncat('There are', length(numericVars), 'Nummeric variables')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:31:56.893622Z","iopub.execute_input":"2024-12-07T08:31:56.895208Z","iopub.status.idle":"2024-12-07T08:31:57.150079Z","shell.execute_reply":"2024-12-07T08:31:57.148197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig (10,10)\n# Distribution off nummeric features\npar(mfrow = c(2, 4))\nfor (i in numericVarNames){\n    hist(df_train[,i],breaks=15,freq=FALSE, main=colnames(df_train[i]), xlab=colnames(df_train[i]))\n    qqnorm( (df_train[,i]), pch = 1, frame = FALSE)\n    qqline( (df_train[,i]), col = \"steelblue\", lwd = 2)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:32:01.019221Z","iopub.execute_input":"2024-12-07T08:32:01.020768Z","iopub.status.idle":"2024-12-07T08:36:22.409228Z","shell.execute_reply":"2024-12-07T08:36:22.407125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_numVar <- all[, numericVars]\ncat('Features with correlation above > 0,3')\ncor_numVar <- cor(all_numVar, use=\"pairwise.complete.obs\") #correlations of all numeric variables\nfindCorrelation(cor_numVar, cutoff = 0.3, verbose = T, names=T)\ncorrplot(cor_numVar, method = 'square', order = 'FPC', type = 'lower', diag = FALSE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:37:35.888124Z","iopub.execute_input":"2024-12-07T08:37:35.889814Z","iopub.status.idle":"2024-12-07T08:37:36.596629Z","shell.execute_reply":"2024-12-07T08:37:36.594762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nall_numVar=all_numVar[complete.cases(all_numVar),] \nM=cor (all_numVar)\npheatmap(M, scale = \"row\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:38:06.815843Z","iopub.execute_input":"2024-12-07T08:38:06.81745Z","iopub.status.idle":"2024-12-07T08:38:07.225414Z","shell.execute_reply":"2024-12-07T08:38:07.223509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_numVar=all_numVar[complete.cases(all_numVar),] \npca <- prcomp(all_numVar, scale. = T, center=T)\npca$rotation[1:8,1:4]\n## make a scree plot\npca.var <- pca$sdev^2\npca.var.per <- round(pca.var/sum(pca.var)*100, 1)\nbarplot(pca.var.per, main=\"Scree Plot\", xlab=\"Principal Component\", ylab=\"Percent Variation\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:38:18.209114Z","iopub.execute_input":"2024-12-07T08:38:18.210685Z","iopub.status.idle":"2024-12-07T08:38:18.845409Z","shell.execute_reply":"2024-12-07T08:38:18.843514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plotten van component loadings\nbiplot(pca,choices=1:2, scale = 1,\ncol = c('white', 'red'))\nbiplot(pca,choices=3:4, scale = 0,\ncol = c('white', 'red'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:38:33.984535Z","iopub.execute_input":"2024-12-07T08:38:33.986192Z","iopub.status.idle":"2024-12-07T08:40:29.556115Z","shell.execute_reply":"2024-12-07T08:40:29.553371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"caret::nearZeroVar(df_train, saveMetrics = TRUE) %>% \n  tibble::rownames_to_column() %>% \n  filter(nzv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:45:19.235558Z","iopub.execute_input":"2024-12-07T08:45:19.237172Z","iopub.status.idle":"2024-12-07T08:46:30.834579Z","shell.execute_reply":"2024-12-07T08:46:30.832501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Looking for non-linear relations**","metadata":{}},{"cell_type":"code","source":"Vis=sample_n(df_train[complete.cases(df_train),], 1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:46:41.498616Z","iopub.execute_input":"2024-12-07T08:46:41.500185Z","iopub.status.idle":"2024-12-07T08:46:41.806641Z","shell.execute_reply":"2024-12-07T08:46:41.804835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age\ncat ('Age:')\nggplot(Vis, aes(age, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from age is:', round(summary(lm(premium_amount~age, data=df_train))$r.squared, digit=3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:46:45.890147Z","iopub.execute_input":"2024-12-07T08:46:45.891838Z","iopub.status.idle":"2024-12-07T08:46:47.381439Z","shell.execute_reply":"2024-12-07T08:46:47.379389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# annual_income \ncat ('annual_income :')\nggplot(Vis, aes(annual_income, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from premium_amount~annual_income is:', round(summary(lm(premium_amount~annual_income, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:47:02.471396Z","iopub.execute_input":"2024-12-07T08:47:02.473077Z","iopub.status.idle":"2024-12-07T08:47:03.585469Z","shell.execute_reply":"2024-12-07T08:47:03.582623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# number_of_dependents\ncat ('number_of_dependents :')\nggplot(Vis, aes(number_of_dependents, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from premium_amount~annual_income is:', round(summary(lm(premium_amount~number_of_dependents, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:47:11.197322Z","iopub.execute_input":"2024-12-07T08:47:11.198971Z","iopub.status.idle":"2024-12-07T08:47:12.40782Z","shell.execute_reply":"2024-12-07T08:47:12.405044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# health_score\ncat ('health_score :')\nggplot(Vis, aes(health_score, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from premium_amount~annual_income is:', round(summary(lm(premium_amount~health_score, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:47:21.415531Z","iopub.execute_input":"2024-12-07T08:47:21.417227Z","iopub.status.idle":"2024-12-07T08:47:22.451402Z","shell.execute_reply":"2024-12-07T08:47:22.448464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# previous_claims\ncat ('previous_claims :')\nggplot(Vis, aes(previous_claims, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from previous_claims is:', round(summary(lm(premium_amount~previous_claims, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:47:45.090561Z","iopub.execute_input":"2024-12-07T08:47:45.092225Z","iopub.status.idle":"2024-12-07T08:47:45.894373Z","shell.execute_reply":"2024-12-07T08:47:45.891422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# vehicle_age\ncat ('vehicle_age :')\nggplot(Vis, aes(vehicle_age, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from vehicle_age is:', round(summary(lm(premium_amount~vehicle_age, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:47:53.969933Z","iopub.execute_input":"2024-12-07T08:47:53.971551Z","iopub.status.idle":"2024-12-07T08:47:55.082618Z","shell.execute_reply":"2024-12-07T08:47:55.079827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# credit_score\ncat ('credit_score  :')\nggplot(Vis, aes(credit_score, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from credit_score is:', round(summary(lm(premium_amount~credit_score, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:48:17.666444Z","iopub.execute_input":"2024-12-07T08:48:17.66816Z","iopub.status.idle":"2024-12-07T08:48:19.104567Z","shell.execute_reply":"2024-12-07T08:48:19.102504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# insurance_duration\ncat ('insurance_duration  :')\nggplot(Vis, aes(insurance_duration, premium_amount))  +\n  geom_point() +\n  stat_smooth()\ncat('The R(2)from insurance_duration is:', round(summary(lm(insurance_duration~credit_score, data=df_train))$r.squared, digit=5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:48:28.274282Z","iopub.execute_input":"2024-12-07T08:48:28.275942Z","iopub.status.idle":"2024-12-07T08:48:29.158954Z","shell.execute_reply":"2024-12-07T08:48:29.156524Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Umap nummeric features**","metadata":{}},{"cell_type":"code","source":"X_UMAP=Vis\nX_UMAP_1=setdiff(colnames(X_UMAP), c('Id', 'premium_amount'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:48:39.437916Z","iopub.execute_input":"2024-12-07T08:48:39.439502Z","iopub.status.idle":"2024-12-07T08:48:39.454707Z","shell.execute_reply":"2024-12-07T08:48:39.452878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Umap overall)\n# Umap\nlibrary(umap)\nset.seed(142)\numap_fit <- X_UMAP[X_UMAP_1] %>%dplyr::select(where(is.numeric)) %>%scale() %>% \n  umap()\numap_df <- umap_fit$layout %>%\n  as.data.frame()%>%\n  rename(UMAP1=\"V1\",\n         UMAP2=\"V2\")\numap_df$Target_klasse=case_when (X_UMAP$premium_amount < 500 ~ 'a.< 500',\n          X_UMAP$premium_amount >= 500 & X_UMAP$premium_amount <1000~ 'b. 500-1000',\n          X_UMAP$premium_amount >= 1000 & X_UMAP$premium_amount <2000~ 'b. 1000-2000',\n          X_UMAP$premium_amount >= 2000 & X_UMAP$premium_amount <3000~ 'c. 2000-3000',\n         X_UMAP$premium_amount >= 3000 & X_UMAP$premium_amount <4000~ 'd. 3000-4000',\n         X_UMAP$premium_amount >= 4000 & X_UMAP$premium_amount <10000~ 'e. 4000+',\n                                  TRUE ~ 'f.Onbekend')\n\np <- ggplot(umap_df, aes(UMAP1, UMAP2))\np + geom_point(aes(colour =Target_klasse))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:48:51.324506Z","iopub.execute_input":"2024-12-07T08:48:51.326138Z","iopub.status.idle":"2024-12-07T08:48:56.182265Z","shell.execute_reply":"2024-12-07T08:48:56.180073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**EDA string features**","metadata":{}},{"cell_type":"code","source":"# Identificeren van onderlinge afhankelijkheid string variabelen.\nall_c=subset( df_train, select = -c(id, premium_amount) )\ncharVars <- which(sapply(all_c, is.character)) #index vector numeric variables\ncharVarNames <- names(charVars) #saving names vector for use later on\ncat('There are', length(charVars), 'character variables')\nall_charVar <- all[, charVars]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:49:11.593332Z","iopub.execute_input":"2024-12-07T08:49:11.595012Z","iopub.status.idle":"2024-12-07T08:49:11.8179Z","shell.execute_reply":"2024-12-07T08:49:11.816092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig (5,5)\n\nfor (i in charVarNames){\nprint (Vis%>%ggplot(aes(x=Vis[,i], y=premium_amount)) + \n  geom_violin()+ coord_flip()+ stat_summary(fun=mean, geom=\"point\", shape=23, size=7)+ ggtitle(i)+xlab (i)+ylab(\"premium amount\") + geom_boxplot(width=0.1))}\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:49:19.598468Z","iopub.execute_input":"2024-12-07T08:49:19.600092Z","iopub.status.idle":"2024-12-07T08:49:23.589139Z","shell.execute_reply":"2024-12-07T08:49:23.587106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# First insides","metadata":{}},{"cell_type":"code","source":"fig (10,10)\ndata_RF=sample_n(df_train[complete.cases(df_train),], 10000)\nquick_RF <- randomForest(premium_amount+1~.-id, data = data_RF, method='anova')\nimp_RF <- importance(quick_RF)\nimp_DF <- data.frame(Variables = row.names(imp_RF), MSE = imp_RF[,1])\nimp_DF <- imp_DF[order(imp_DF$MSE, decreasing = TRUE),]\n\nggplot(imp_DF[1:20,], aes(x=reorder(Variables, MSE), y=MSE, fill=MSE)) + geom_bar(stat = 'identity') + labs(x = 'Variables', y= '% increase MSE if variable is randomly permuted') + coord_flip() + theme(legend.position=\"none\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:49:36.263566Z","iopub.execute_input":"2024-12-07T08:49:36.265259Z","iopub.status.idle":"2024-12-07T08:51:00.662421Z","shell.execute_reply":"2024-12-07T08:51:00.659942Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Recipes + Modelling\nLet's try some modeling","metadata":{}},{"cell_type":"markdown","source":"# Lasso regression","metadata":{}},{"cell_type":"code","source":"#validation_split <- validation_split(df.train, prop = 0.8)\nvalidation_split <- vfold_cv(df.train, v = 5, strata = premium_amount)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:51:17.786145Z","iopub.execute_input":"2024-12-07T08:51:17.7877Z","iopub.status.idle":"2024-12-07T08:51:18.92367Z","shell.execute_reply":"2024-12-07T08:51:18.921742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"elastic_model <- linear_reg(penalty = tune(), mixture = tune()) %>%\n  set_engine(\"glmnet\")\nparam_grid <- grid_regular(penalty(), \n                            mixture(),\n                            levels = list(penalty = 25,\n                                          mixture = 10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:52:05.987022Z","iopub.execute_input":"2024-12-07T08:52:05.988643Z","iopub.status.idle":"2024-12-07T08:52:06.020916Z","shell.execute_reply":"2024-12-07T08:52:06.019138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Recipes\nblueprint=recipe(premium_amount~., data=df_train)%>%\nupdate_role(id,new_role=\"id variable\")%>%\nstep_other(all_nominal_predictors(), threshold=0.01)%>%\nstep_date(policy_start_date, label = TRUE,   features = c(\"dow\", \"month\", \"week\", \"year\"), keep_original_cols = FALSE) %>% \nstep_mutate(ouderdom=2024-policy_start_date_year)%>%\nstep_string2factor(all_nominal(), skip = TRUE)%>%\n#step_impute_bag(all_predictors())%>%\n step_impute_median(all_numeric_predictors()) %>% \nstep_impute_mode(all_nominal_predictors()) %>% \nstep_YeoJohnson(all_numeric_predictors())%>%\nstep_normalize(all_numeric_predictors()) %>%\n#step_center(all_numeric_predictors()) %>%\nstep_scale(all_numeric_predictors()) %>%\nstep_dummy(all_nominal_predictors(), one_hot = TRUE)%>%\n#step_lincomb(all_numeric_predictors())%>%\nstep_nzv(all_predictors(), freq_cut =95/5) %>%\nstep_corr(all_numeric_predictors(), threshold = 0.90) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:53:01.62042Z","iopub.execute_input":"2024-12-07T08:53:01.622063Z","iopub.status.idle":"2024-12-07T08:53:01.888591Z","shell.execute_reply":"2024-12-07T08:53:01.886696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"workflow <- workflow() %>%\n  add_model(elastic_model) %>% \n  add_recipe(blueprint)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:53:10.207325Z","iopub.execute_input":"2024-12-07T08:53:10.208938Z","iopub.status.idle":"2024-12-07T08:53:10.224197Z","shell.execute_reply":"2024-12-07T08:53:10.222392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tune_result <- workflow %>% \n  tune_grid(validation_split,\n            grid = param_grid,\n            metrics = metric_set(yardstick::rmse))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T08:53:15.802499Z","iopub.execute_input":"2024-12-07T08:53:15.804206Z","iopub.status.idle":"2024-12-07T09:14:30.080198Z","shell.execute_reply":"2024-12-07T09:14:30.078269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Tune=tune_result %>% \n  collect_metrics()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:16:54.388974Z","iopub.execute_input":"2024-12-07T09:16:54.390834Z","iopub.status.idle":"2024-12-07T09:16:54.434802Z","shell.execute_reply":"2024-12-07T09:16:54.432861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tune_result%>%show_best ()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:07.41728Z","iopub.execute_input":"2024-12-07T09:17:07.418836Z","iopub.status.idle":"2024-12-07T09:17:07.556718Z","shell.execute_reply":"2024-12-07T09:17:07.554724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(plotly)\nfig <- plot_ly(Tune, x =~penalty, y = ~mixture, z = ~mean, type = \"contour\",\n             width = 2, height = 2)\n\nfig","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:19.626637Z","iopub.execute_input":"2024-12-07T09:17:19.628488Z","iopub.status.idle":"2024-12-07T09:17:20.807181Z","shell.execute_reply":"2024-12-07T09:17:20.804735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tune_best=tune_result%>%select_best ()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:30.203469Z","iopub.execute_input":"2024-12-07T09:17:30.205337Z","iopub.status.idle":"2024-12-07T09:17:30.294306Z","shell.execute_reply":"2024-12-07T09:17:30.292156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"uit_elastic_model <- \n    linear_reg(penalty = tune_best$penalty,\n               mixture = tune_best$mixture) %>%\n    set_engine(\"glmnet\")\n\n# Workflow opstellen\nELAS_wflow=workflow()%>%\nadd_model(uit_elastic_model)%>%\nadd_recipe(blueprint)\n\nELAS_fit=ELAS_wflow%>%fit(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:41.392894Z","iopub.execute_input":"2024-12-07T09:17:41.394714Z","iopub.status.idle":"2024-12-07T09:19:32.592834Z","shell.execute_reply":"2024-12-07T09:19:32.590912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Var_lijst_ELAS=ELAS_fit%>%tidy()%>%mutate (estimate_1=abs (estimate))%>%arrange(desc(estimate_1))\nggplot(Var_lijst_ELAS[2:20,], aes(x=reorder(term, estimate), y=estimate, fill=estimate_1)) + geom_bar(stat = 'identity') + labs(x = 'Variables', y= 'Gestand. waarde parameter') + coord_flip() + theme(legend.position=\"none\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:19:41.569717Z","iopub.execute_input":"2024-12-07T09:19:41.571328Z","iopub.status.idle":"2024-12-07T09:19:41.871179Z","shell.execute_reply":"2024-12-07T09:19:41.869173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"P_ELAS=predict (ELAS_fit, new_data=df_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:20:04.769553Z","iopub.execute_input":"2024-12-07T09:20:04.771325Z","iopub.status.idle":"2024-12-07T09:20:09.666442Z","shell.execute_reply":"2024-12-07T09:20:09.664585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Grafische vergelijk predicted value en werkelijke waarde\nVergelijk_ELAS=as.data.frame (cbind (df_val$premium_amount,P_ELAS$.pred))\ncolnames (Vergelijk_ELAS)=c('Werkelijk', 'Schatting')\nggplot(Vergelijk_ELAS, aes(x = Werkelijk, y=Schatting)) + \n  # Create a diagonal line:\n  geom_abline(lty = 2) + \n  geom_point(alpha = 0.5) + \n  labs(y = \"Schatting\", x = \"Werkelijk\") +\n  # Scale and size the x- and y-axis uniformly:\n  coord_obs_pred()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:21:00.580238Z","iopub.execute_input":"2024-12-07T09:21:00.581807Z","iopub.status.idle":"2024-12-07T09:21:11.628822Z","shell.execute_reply":"2024-12-07T09:21:11.626858Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GLM with penalizer in H2O","metadata":{}},{"cell_type":"code","source":"# Transform to orginal target for GLM\ndf_train$premium_amount=exp(df_train$premium_amount)\n# Recipes\nGLM_recipe=recipe(premium_amount~., data=df_train)%>%\nupdate_role(id,new_role=\"id variable\")%>%\nstep_other(all_nominal_predictors(), threshold=0.01)%>%\nstep_date(policy_start_date, label = TRUE,   features = c(\"dow\", \"month\", \"week\", \"year\"), keep_original_cols = FALSE) %>% \nstep_mutate(ouderdom=2024-policy_start_date_year)%>%\nstep_string2factor(all_nominal(), skip = TRUE)%>%\n#step_impute_bag(all_predictors())%>%\nstep_impute_median(all_numeric_predictors()) %>% \nstep_impute_mode(all_nominal_predictors()) %>% \nstep_YeoJohnson(all_numeric_predictors())%>%\nstep_normalize(all_numeric_predictors()) %>%\n#step_center(all_numeric_predictors()) %>%\nstep_scale(all_numeric_predictors()) %>%\nstep_dummy(all_nominal_predictors(), one_hot = TRUE)%>%\n#step_lincomb(all_numeric_predictors())%>%\nstep_nzv(all_predictors(), freq_cut =95/5) %>%\nstep_corr(all_numeric_predictors(), threshold = 0.90) %>%\nstep_novel(all_predictors(), -all_numeric())\nh2o.init()\ntrain_h2o <- prep(GLM_recipe, training = df_train, retain = TRUE) %>%\n  juice() %>%\n  as.h2o()\ntest_h2o <- prep(GLM_recipe, training = df_train) %>%\n  bake(new_data = df_val) %>%\n  as.h2o()\nscore_h2o <- prep(GLM_recipe, training = df_train) %>%\n  bake(new_data = df.test) %>%\n  as.h2o()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:08:54.829939Z","iopub.execute_input":"2024-12-08T14:08:54.831532Z","iopub.status.idle":"2024-12-08T14:14:50.049464Z","shell.execute_reply":"2024-12-08T14:14:50.047618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y <- \"premium_amount\"\nX <- setdiff(names(train_h2o [2:59]), c(Y))\nbest_glm <- h2o.glm(\n  x = X, y = Y, training_frame = train_h2o, alpha = 0, family =c(\"gamma\"),link =\"log\", lambda_search=TRUE, \n  nfolds = 10, fold_assignment = \"Modulo\", stopping_metric=\"RMSLE\",\n  keep_cross_validation_predictions = TRUE, seed = 123, max_iterations_dispersion = 300\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:16:51.529135Z","iopub.execute_input":"2024-12-08T14:16:51.530708Z","iopub.status.idle":"2024-12-08T14:22:09.755832Z","shell.execute_reply":"2024-12-08T14:22:09.754081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"h2o.performance (best_glm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:25:32.549862Z","iopub.execute_input":"2024-12-08T14:25:32.551842Z","iopub.status.idle":"2024-12-08T14:25:32.576078Z","shell.execute_reply":"2024-12-08T14:25:32.574333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"coef_glm=as.data.frame(h2o.coef(best_glm))\ndim(coef_glm)\ncolnames(coef_glm)=c(\"waarde\")\nsubset (coef_glm, waarde != 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:25:52.437019Z","iopub.execute_input":"2024-12-08T14:25:52.438554Z","iopub.status.idle":"2024-12-08T14:25:52.497176Z","shell.execute_reply":"2024-12-08T14:25:52.495413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"h2o.varimp_plot(best_glm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:26:05.027512Z","iopub.execute_input":"2024-12-08T14:26:05.029195Z","iopub.status.idle":"2024-12-08T14:26:05.290078Z","shell.execute_reply":"2024-12-08T14:26:05.288258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred <- h2o.predict(object=best_glm, newdata=test_h2o)\nlibrary (Metrics)\nrmsle(pred, exp(df_val$premium_amount))\npred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:28:49.492744Z","iopub.execute_input":"2024-12-08T14:28:49.494309Z","iopub.status.idle":"2024-12-08T14:28:50.488478Z","shell.execute_reply":"2024-12-08T14:28:50.486649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# AUTO ML in H2O","metadata":{}},{"cell_type":"code","source":"df_train$premium_amount=log (df_train$premium_amount)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Modeling packages\nh2o.init()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Oorspronkelijke features\nsample=sample_n(df_train, 200000)\ndf.train.h2o <- as.h2o (sample)\ndf.val.h2o <- as.h2o(df_val)\n# test data\ndf.test.h2o <- as.h2o(df.test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# run for a maximum of 1200 seconds\nauto_mod <- auto_ml() %>%\n  set_engine(\"h2o\",\n    max_models = 30,\n    stopping_metric = \"rmse\",\n    sort_metric = \"rmse\",\n    balance_classes = T,\n    seed = 42\n  ) %>%\n  set_mode(\"regression\") %>%\n  translate()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Recipes\nauto_recipe=recipe(premium_amount~., data=sample)%>%\nupdate_role(id,new_role=\"id variable\")%>%\n##step_other(all_nominal_predictors(), threshold=0.01)%>%\nstep_date(policy_start_date, label = TRUE,   features = c(\"dow\", \"month\", \"week\", \"year\"), keep_original_cols = FALSE) %>% \n#step_mutate(ouderdom=2024-policy_start_date_year)%>%\nstep_string2factor(all_nominal(), skip = TRUE)%>%\n#step_impute_bag(all_predictors())%>%\n step_impute_median(all_numeric_predictors()) %>% \nstep_impute_mode(all_nominal_predictors()) %>% \n#step_YeoJohnson(all_numeric_predictors())%>%\nstep_normalize(all_numeric_predictors()) %>%\n#step_center(all_numeric_predictors()) %>%\nstep_scale(all_numeric_predictors()) %>%\nstep_dummy(all_nominal_predictors(), one_hot = TRUE)%>%\n#step_lincomb(all_numeric_predictors())%>%\nstep_nzv(all_predictors(), freq_cut =95/5) %>%\n#step_corr(all_numeric_predictors(), threshold = 0.90) %>%\nstep_novel(all_predictors(), -all_numeric())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"auto_workflow <-\n  workflow() %>%\n  add_model(auto_mod) %>%\n  add_recipe(auto_recipe)\n\nauto_res <-\n  auto_workflow %>%\n  fit(data = sample)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head (agua::rank_results(auto_res)%>%filter (.metric==\"rmse\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head (collect_metrics(auto_res, summarize = FALSE))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"auto_res %>%\n  extract_fit_parsnip() %>%\n  member_weights() %>%\n  unnest(importance) %>%\n  filter(type == \"scaled_importance\") %>%\n  ggplot() +\n  geom_boxplot(aes(value, algorithm)) +\n  scale_x_sqrt() +\n  labs(y = NULL, x = \"scaled importance\", title = \"Member importance in stacked ensembles\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX_val=df_val %>% dplyr::select( -premium_amount)\npred_sl = predict(auto_res, new_data = X_val)\npred_sl=exp(pred_sl)\\\nval=as.data.frame(cbind (df_val$premium_amount, pred_sl))\ncolnames (val)=c('werkelijk', 'voorspelling')\nMetrics::rmse(val$werkelijk, val$voorspelling)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# GLM submission\nid = as.data.frame(df.test$id)\n#pred_tl = predict(auto_res, new_data = X_test)_tl\n#pred_tl=exp(pred_tl)\\\n#pred_tl <- as.data.frame(h2o.predict(object=best_glm, newdata=score_h2o))\n#sub <- cbind (id, pred_tl$predict)\n#colnames (sub)=c('id', 'premium amount')\n#write_csv(sub, \"submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:41:09.279368Z","iopub.execute_input":"2024-12-08T14:41:09.280979Z","iopub.status.idle":"2024-12-08T14:41:11.862378Z","shell.execute_reply":"2024-12-08T14:41:11.860543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Submission\nid = df.test$id\nX_score=df.test\npred_sl = predict(auto_res, new_data = X_score)\npred_tl=exp(pred_tl)\nsub <- as.data.frame(cbind (id, pred_tl))\ncolnames (sub)=c('id', 'premium amount')\nwrite_csv(sub, \"submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}