{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.4.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30749,"isInternetEnabled":true,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This R environment comes with many helpful analytics packages installed\n# It is defined by the kaggle/rstats Docker image: https://github.com/kaggle/docker-rstats\n# For example, here's a helpful package to load\n\nlibrary(tidyverse) # metapackage of all tidyverse packages\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nlist.files(path = \"../input\")\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:20:59.673539Z","iopub.execute_input":"2024-12-30T19:20:59.675522Z","iopub.status.idle":"2024-12-30T19:21:01.006046Z","shell.execute_reply":"2024-12-30T19:21:01.004256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"suppressPackageStartupMessages({\nlibrary(MASS)\nlibrary(stats)        \nlibrary(car) \nlibrary(corrplot) \nlibrary(naniar)\nlibrary(randomForest)\nlibrary(xgboost)\nlibrary(e1071)\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:01.008856Z","iopub.execute_input":"2024-12-30T19:21:01.049199Z","iopub.status.idle":"2024-12-30T19:21:02.636098Z","shell.execute_reply":"2024-12-30T19:21:02.634302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train <- read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test <- read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:02.638845Z","iopub.execute_input":"2024-12-30T19:21:02.640605Z","iopub.status.idle":"2024-12-30T19:21:11.381416Z","shell.execute_reply":"2024-12-30T19:21:11.379693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:11.383897Z","iopub.execute_input":"2024-12-30T19:21:11.385225Z","iopub.status.idle":"2024-12-30T19:21:11.442258Z","shell.execute_reply":"2024-12-30T19:21:11.44041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(df_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:11.444877Z","iopub.execute_input":"2024-12-30T19:21:11.446372Z","iopub.status.idle":"2024-12-30T19:21:11.485558Z","shell.execute_reply":"2024-12-30T19:21:11.483781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_numeric <- df_train\nfor (col in names(df_train_numeric)) {\n  if (is.character(df_train_numeric[[col]])) {\n    df_train_numeric[[col]] <- as.numeric(factor(df_train_numeric[[col]], exclude = NULL))\n  }\n  \n  if (is.factor(df_train_numeric[[col]])) {\n    df_train_numeric[[col]] <- as.numeric(df_train_numeric[[col]])\n  }\n  \n  if (inherits(df_train_numeric[[col]], \"POSIXct\")) {\n    df_train_numeric[[col]] <- as.numeric(df_train_numeric[[col]])\n  }\n}\n\nstr(df_train_numeric)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:11.488289Z","iopub.execute_input":"2024-12-30T19:21:11.489841Z","iopub.status.idle":"2024-12-30T19:21:12.57736Z","shell.execute_reply":"2024-12-30T19:21:12.575599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(df_train_numeric)\nsummary(df_train_numeric)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:12.580176Z","iopub.execute_input":"2024-12-30T19:21:12.581667Z","iopub.status.idle":"2024-12-30T19:21:13.872829Z","shell.execute_reply":"2024-12-30T19:21:13.871017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(df_test,1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:13.875385Z","iopub.execute_input":"2024-12-30T19:21:13.876789Z","iopub.status.idle":"2024-12-30T19:21:13.921695Z","shell.execute_reply":"2024-12-30T19:21:13.91991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:13.924294Z","iopub.execute_input":"2024-12-30T19:21:13.925684Z","iopub.status.idle":"2024-12-30T19:21:13.959276Z","shell.execute_reply":"2024-12-30T19:21:13.957523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Karakter değişkenleri otomatik tespiti\nchar_cols <- sapply(df_train, is.character)\n\n# Karakter değişkenlerini faktöre, ardından nümeriğe dönüştürme\nfor (col in names(df_train)[char_cols]) {\n  df_train[[col]] <- as.numeric(factor(df_train[[col]], exclude = NULL))\n}\n\n# Sonuçları kontrol edin\nstr(df_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:21:13.961875Z","iopub.execute_input":"2024-12-30T19:21:13.963314Z","iopub.status.idle":"2024-12-30T19:21:14.683786Z","shell.execute_reply":"2024-12-30T19:21:14.681924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_train ve df_test'in sayısal sütunlarını standartlaştır\nstandardize <- function(data) {\n  numeric_cols <- sapply(data, is.numeric) # Sayısal sütunları bul\n  data[numeric_cols] <- scale(data[numeric_cols]) # scale ile standartlaştır\n  return(data)\n}\n\n# Standartlaştırılmış veri çerçevelerini oluştur\ndf_train <- standardize(df_train)\ndf_test <- standardize(df_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:26:27.542903Z","iopub.execute_input":"2024-12-30T19:26:27.544785Z","iopub.status.idle":"2024-12-30T19:26:33.362423Z","shell.execute_reply":"2024-12-30T19:26:33.360514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"char_cols <- sapply(df_test, is.character)\n\n# Karakter değişkenlerini faktöre, ardından nümeriğe dönüştürme\nfor (col in names(df_test)[char_cols]) {\n  df_test[[col]] <- as.numeric(factor(df_test[[col]], exclude = NULL))\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:26:37.738632Z","iopub.execute_input":"2024-12-30T19:26:37.740309Z","iopub.status.idle":"2024-12-30T19:26:37.760794Z","shell.execute_reply":"2024-12-30T19:26:37.75905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_train_cleaned oluşturma\ndf_train_cleaned <- df_train[, !(names(df_train) %in% c(\"id\", \"Policy Start Date\"))]\n\n# df_test_cleaned oluşturma\ndf_test_cleaned <- df_test[, !(names(df_test) %in% c(\"id\", \"Policy Start Date\"))]\n\n# Sonuçları kontrol etme\nstr(df_train_cleaned)\nstr(df_test_cleaned)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:26:43.223225Z","iopub.execute_input":"2024-12-30T19:26:43.22481Z","iopub.status.idle":"2024-12-30T19:26:43.270232Z","shell.execute_reply":"2024-12-30T19:26:43.268447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 'Premium Amount' hedef değişken (y) olarak ayırma\ny <- df_train_cleaned$`Premium Amount`\n\n# Özellikleri (x) ayırma\nx <- df_train_cleaned[, !(names(df_train_cleaned) %in% c(\"Premium Amount\"))]\n\n# Eğitim ve test setlerine ayırmak için 'caret' veya 'base' fonksiyonları kullanılabilir\n# Burada 'caret' kütüphanesini kullanıyoruz\nlibrary(caret)\n\nset.seed(42) # Rastgeleliği kontrol etmek için\ntrain_index <- createDataPartition(y, p = 0.8, list = FALSE)\n\nx_train <- x[train_index, ]\nx_test <- x[-train_index, ]\n\ny_train <- y[train_index]\ny_test <- y[-train_index]\n\n# Sonuçları kontrol etme\nstr(x_train)\nstr(x_test)\nstr(y_train)\nstr(y_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:26:53.588598Z","iopub.execute_input":"2024-12-30T19:26:53.590295Z","iopub.status.idle":"2024-12-30T19:26:55.056205Z","shell.execute_reply":"2024-12-30T19:26:55.054508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(Metrics)\nx_train_matrix <- as.matrix(x_train)\nx_test_matrix <- as.matrix(x_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:27:05.230828Z","iopub.execute_input":"2024-12-30T19:27:05.232474Z","iopub.status.idle":"2024-12-30T19:27:05.413587Z","shell.execute_reply":"2024-12-30T19:27:05.411684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"####\nlibrary(lightgbm)\ndf_train <- lgb.Dataset(data = x_train_matrix, label = y_train)\nparams <- list(\n  objective = \"regression\", \n  metric = \"rmse\",          \n  num_leaves = 31,          \n  learning_rate = 0.1       \n)\n\nset.seed(42)\nlightgbm_model <- lgb.train(\n  params = params,\n  data = df_train,\n  nrounds = 28, # İterasyon \n  verbose = 0\n)\n\n####\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:25.202652Z","iopub.execute_input":"2024-12-30T20:05:25.20429Z","iopub.status.idle":"2024-12-30T20:05:27.394128Z","shell.execute_reply":"2024-12-30T20:05:27.391875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test[is.na(y_test)] <- median(y_test, na.rm = TRUE)\ny_test[y_test < 0] <- 0\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:32.47299Z","iopub.execute_input":"2024-12-30T20:05:32.474617Z","iopub.status.idle":"2024-12-30T20:05:32.496959Z","shell.execute_reply":"2024-12-30T20:05:32.495069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"log_rmse <- sqrt(mean((log1p(y_test) - log1p(y_pred))^2))\nprint(paste(\"LightGBM Log RMSE:\", log_rmse))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:36.695119Z","iopub.execute_input":"2024-12-30T20:05:36.696805Z","iopub.status.idle":"2024-12-30T20:05:36.720252Z","shell.execute_reply":"2024-12-30T20:05:36.71844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(log_rmse)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:38.805873Z","iopub.execute_input":"2024-12-30T20:05:38.807473Z","iopub.status.idle":"2024-12-30T20:05:38.82095Z","shell.execute_reply":"2024-12-30T20:05:38.819138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\npar(mfrow = c(1, 2)) \nhist(\n  y_pred,\n  breaks = 30, \n  col = \"blue\", \n  main = \"Histogram of y_pred\",\n  xlab = \"Predicted Values\",\n  ylab = \"Frequency\"\n)\n\n\nhist(\n  y_test,\n  breaks = 30,\n  col = \"green\",\n  main = \"Histogram of y_test\",\n  xlab = \"Actual Values\",\n  ylab = \"Frequency\"\n)\npar(mfrow = c(1, 1))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:44.056998Z","iopub.execute_input":"2024-12-30T20:05:44.058605Z","iopub.status.idle":"2024-12-30T20:05:44.163216Z","shell.execute_reply":"2024-12-30T20:05:44.161311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_subm=('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:51.386952Z","iopub.execute_input":"2024-12-30T20:05:51.388612Z","iopub.status.idle":"2024-12-30T20:05:51.401232Z","shell.execute_reply":"2024-12-30T20:05:51.399276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_cleaned_matrix <- as.matrix(df_test_cleaned)\nlightgbm_model <- predict(xgb_model, df_test_cleaned_matrix)\nhead(lightgbm_model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:05:53.427196Z","iopub.execute_input":"2024-12-30T20:05:53.428778Z","iopub.status.idle":"2024-12-30T20:05:54.061735Z","shell.execute_reply":"2024-12-30T20:05:54.05932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission <- df_test %>%\n  dplyr::select(id) %>% \n  mutate(`Premium Amount` = lightgbm_model) \n\n\nhead(submission)\n\n\nwrite.csv(submission, \"submission.csv\", row.names = FALSE)\n\ncat(\"Submission file saved as 'submission.csv'.\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:06:03.105139Z","iopub.execute_input":"2024-12-30T20:06:03.154945Z","iopub.status.idle":"2024-12-30T20:06:05.7687Z","shell.execute_reply":"2024-12-30T20:06:05.766854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}