{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:55:30.042395Z","iopub.execute_input":"2024-12-19T08:55:30.042859Z","iopub.status.idle":"2024-12-19T08:55:30.425819Z","shell.execute_reply.started":"2024-12-19T08:55:30.042822Z","shell.execute_reply":"2024-12-19T08:55:30.424599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# When experimenting with multiple models cv's - score is around 1.04\nThis notebook is extension of my earlier notebook (https://www.kaggle.com/code/pradeep13/insurance-competition-learn-feature-rank-first); moving to modelling following laws of data science, here, following basics and attempting to model from my previous learnings and discussions","metadata":{}},{"cell_type":"code","source":"import subprocess\nimport sys\n\ndef install_and_import(libraries):\n    for library_name in libraries:\n        try:\n            __import__(library_name)\n            print(f\"{library_name} is already installed.\")\n        except ImportError:\n            print(f\"{library_name} not found. Installing...\")\n            subprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", library_name])\n\n# Example usage:\nlibraries = [\"polars\", \"category_encoders\", \"lightgbm\", \"xgboost\", \"scikit-learn==1.2.2\", \"scipy\"]\ninstall_and_import(libraries)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:55:35.562246Z","iopub.execute_input":"2024-12-19T08:55:35.56286Z","iopub.status.idle":"2024-12-19T08:55:41.255906Z","shell.execute_reply.started":"2024-12-19T08:55:35.562813Z","shell.execute_reply":"2024-12-19T08:55:41.254833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load all libraries required","metadata":{}},{"cell_type":"code","source":"import category_encoders as ce\nimport lightgbm as lgb\nimport numpy as np\nimport os\nimport pandas as pd\nimport polars as pl\nimport xgboost as xgb\n\nfrom xgboost import cv\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import cross_val_score, KFold\nfrom sklearn.metrics import mean_squared_log_error, make_scorer, mean_squared_error\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:07.491246Z","iopub.execute_input":"2024-12-19T08:58:07.491729Z","iopub.status.idle":"2024-12-19T08:58:07.498138Z","shell.execute_reply.started":"2024-12-19T08:58:07.491693Z","shell.execute_reply":"2024-12-19T08:58:07.496827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\nins_train = pl.read_csv('/kaggle/input/playground-series-s4e12/train.csv', try_parse_dates=True)\nins_test = pl.read_csv('/kaggle/input/playground-series-s4e12/test.csv', try_parse_dates=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:56:20.614043Z","iopub.execute_input":"2024-12-19T08:56:20.61442Z","iopub.status.idle":"2024-12-19T08:56:23.79143Z","shell.execute_reply.started":"2024-12-19T08:56:20.614387Z","shell.execute_reply":"2024-12-19T08:56:23.790311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# From earlier notebook experiments and EDA, we know that categorical are balanced against target and numerical have missing and XGboost cannot alone improve the model accuracy.","metadata":{}},{"cell_type":"markdown","source":"# Detailed EDA of this data is availble in below notebook\nhttps://www.kaggle.com/code/pradeep13/predict-insurance-premium-eda-simulated-values","metadata":{}},{"cell_type":"markdown","source":"# Hence, used Policy Start Date to generate Day, Week, Month and Year of the Policy Start Date as new numerical feature here. Also, one can observe later Year is good predictor.","metadata":{}},{"cell_type":"code","source":"# create new features from policy start date colum\nins_train = ins_train.with_columns(\n    pl.col(\"Policy Start Date\").dt.day().alias(\"DayofPolicy\"),\n    pl.col(\"Policy Start Date\").dt.week().alias(\"WeekofPolicy\"),\n    pl.col(\"Policy Start Date\").dt.month().alias(\"MonthofPolicy\"),\n    pl.col(\"Policy Start Date\").dt.year().alias(\"YearofPolicy\")\n)\n\nins_test = ins_test.with_columns(\n    pl.col(\"Policy Start Date\").dt.day().alias(\"DayofPolicy\"),\n    pl.col(\"Policy Start Date\").dt.week().alias(\"WeekofPolicy\"),\n    pl.col(\"Policy Start Date\").dt.month().alias(\"MonthofPolicy\"),\n    pl.col(\"Policy Start Date\").dt.year().alias(\"YearofPolicy\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:56:49.846362Z","iopub.execute_input":"2024-12-19T08:56:49.846877Z","iopub.status.idle":"2024-12-19T08:56:49.968622Z","shell.execute_reply.started":"2024-12-19T08:56:49.846842Z","shell.execute_reply":"2024-12-19T08:56:49.966883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Divide column types\ncategorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation',  'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n\nnumeric_columns = ['Age', 'Annual Income', 'Number of Dependents', 'Vehicle Age', 'Credit Score', 'Insurance Duration', 'DayofPolicy', 'WeekofPolicy', 'MonthofPolicy', 'YearofPolicy']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:57:05.81456Z","iopub.execute_input":"2024-12-19T08:57:05.815013Z","iopub.status.idle":"2024-12-19T08:57:05.820675Z","shell.execute_reply.started":"2024-12-19T08:57:05.814977Z","shell.execute_reply":"2024-12-19T08:57:05.819174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# First pre-process categorical variabels","metadata":{}},{"cell_type":"code","source":"# Preprocessing categorical variables\n# Below is the function used to label encode categorical columns \ndef label_encode_column(df: pl.DataFrame, col_name: str) -> pl.Expr: \n    # Get unique values and create a mapping\n    unique_values = df[col_name].unique().sort()\n    mapping = {val: idx for idx, val in enumerate(unique_values)}\n    \n    # Return a Polars expression for mapping\n    return pl.col(col_name).replace_strict(mapping)\n\n# Label encode categorical columns\nfor col in categorical_columns:\n    ins_train = ins_train.with_columns(\n        label_encode_column(ins_train, col).alias(col)\n    )\n\n# Do the same for test data using train data's mapping\nfor col in categorical_columns:\n    ins_test = ins_test.with_columns(\n        label_encode_column(ins_test, col).alias(col)\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:57:11.097834Z","iopub.execute_input":"2024-12-19T08:57:11.098266Z","iopub.status.idle":"2024-12-19T08:57:11.968191Z","shell.execute_reply.started":"2024-12-19T08:57:11.098234Z","shell.execute_reply":"2024-12-19T08:57:11.967009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Next pre-processing Numericals","metadata":{}},{"cell_type":"code","source":"numeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median'))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:57:16.862188Z","iopub.execute_input":"2024-12-19T08:57:16.862676Z","iopub.status.idle":"2024-12-19T08:57:16.869261Z","shell.execute_reply.started":"2024-12-19T08:57:16.862637Z","shell.execute_reply":"2024-12-19T08:57:16.868147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Re- checking \ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='Unknown')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Combine preprocessing steps\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_columns),\n        ('cat', categorical_transformer, categorical_columns)\n    ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:21.045314Z","iopub.execute_input":"2024-12-19T08:58:21.045747Z","iopub.status.idle":"2024-12-19T08:58:21.051548Z","shell.execute_reply.started":"2024-12-19T08:58:21.045715Z","shell.execute_reply":"2024-12-19T08:58:21.050159Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Next create pipeline which will be easy for later use","metadata":{}},{"cell_type":"code","source":"pmmodel = Pipeline([\n    ('preprocessor', preprocessor),\n    ('regressor', xgb.XGBRegressor(\n        random_state=77, \n        tree_method='hist', \n        device='cpu', \n        n_jobs=-1\n    ))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:26.590852Z","iopub.execute_input":"2024-12-19T08:58:26.5912Z","iopub.status.idle":"2024-12-19T08:58:26.597632Z","shell.execute_reply.started":"2024-12-19T08:58:26.591175Z","shell.execute_reply":"2024-12-19T08:58:26.596074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Next is about getting polars data frame ready for pipelines, as polars dataframe still has issues with scikit learn packages from my observations, converting next to pandas dataframe for all modelling purposes","metadata":{}},{"cell_type":"markdown","source":"# Taking log of target and evaluated accordingly","metadata":{}},{"cell_type":"code","source":"# Prepare data\ntrain = ins_train.to_pandas()\ntest = ins_test.to_pandas()\n\n# y = ins_train.select(pl.col(\"Premium Amount\"))\ny = train[\"Premium Amount\"]\ny_log = np.log1p(y)\nX = train.drop([\"id\", \"Premium Amount\", \"Policy Start Date\"], axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:29.569985Z","iopub.execute_input":"2024-12-19T08:58:29.570372Z","iopub.status.idle":"2024-12-19T08:58:30.343535Z","shell.execute_reply.started":"2024-12-19T08:58:29.570344Z","shell.execute_reply":"2024-12-19T08:58:30.342327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# custom rmsle for cross validation, taken cut off as 25 as \n# from my EDA first cut Mode of Premium Amount was 25,\n# other can play upto 100 max, otherwise we might loose a lot\n\ndef custom_rmsle(y_true, y_pred):\n    y_pred = np.maximum(y_pred, 25)\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\nRMSLE = make_scorer(custom_rmsle, greater_is_better=False)\n\n# cross-validation score udf, adapted version\ndef cross_val_score_log(model, X, y, n_splits=5, random_state=77):\n    \n    kf = KFold(n_splits = 5, shuffle=True, random_state=random_state)\n    scores = []\n\n    for train_index, test_index in kf.split(X):\n        X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n        y_train, y_test = y_log.iloc[train_index], y_log.iloc[test_index]\n        \n        model.fit(X_train, y_train)\n        preds = model.predict(X_test)\n        \n        score = np.sqrt(mean_squared_error(y_test, preds))\n        scores.append(score)\n\n    mean_score = np.mean(scores)\n    return mean_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:34.156854Z","iopub.execute_input":"2024-12-19T08:58:34.157265Z","iopub.status.idle":"2024-12-19T08:58:34.165321Z","shell.execute_reply.started":"2024-12-19T08:58:34.157237Z","shell.execute_reply":"2024-12-19T08:58:34.163889Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# As mentioned earlier first let's get start with XGBoost which was tested earlier\nAlso I am using best 5 fold parameters as latest scikit learn 1.6.0 is having some issues in kaggle.\nHence, below code is protype of three models","metadata":{}},{"cell_type":"code","source":"pmmodel_1 = xgb.XGBRegressor(random_state=77, n_jobs=-1, \n                             device='cpu', \n                             n_estimators = 221,\n                             learning_rate = 0.289,\n                             colsample_bytree = 0.828,\n                             max_depth = 8,\n                             min_child_weight = 10,\n                             tree_method = 'hist', \n                             enable_categorical=True,\n                             subsample = 0.795)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:37.585227Z","iopub.execute_input":"2024-12-19T08:58:37.585784Z","iopub.status.idle":"2024-12-19T08:58:37.592223Z","shell.execute_reply.started":"2024-12-19T08:58:37.585736Z","shell.execute_reply":"2024-12-19T08:58:37.590824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pmmodel_2 = lgb.LGBMRegressor(random_state=77, n_jobs=-1, device='cpu',\n                             n_estimators = 600,\n                             learning_rate = 0.055,\n                             max_depth = 12,\n                             verbose = -1,\n                             bagging_fraction = 0.5,\n                             num_leaves = 75,\n                             feature_fraction = 0.99,\n                             min_data_in_lead = 97\n                             )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:40.372594Z","iopub.execute_input":"2024-12-19T08:58:40.37297Z","iopub.status.idle":"2024-12-19T08:58:40.378076Z","shell.execute_reply.started":"2024-12-19T08:58:40.372941Z","shell.execute_reply":"2024-12-19T08:58:40.376629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pmmodel_3 = HistGradientBoostingRegressor(random_state=77, \n                                         max_iter = 437,\n                                         learning_rate = 0.09,\n                                         l2_regularization = 0.02,\n                                         max_leaf_nodes = 100,\n                                         max_depth = 5,\n                                         min_samples_leaf = 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:42.899675Z","iopub.execute_input":"2024-12-19T08:58:42.900092Z","iopub.status.idle":"2024-12-19T08:58:42.905324Z","shell.execute_reply.started":"2024-12-19T08:58:42.900063Z","shell.execute_reply":"2024-12-19T08:58:42.90405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test set\ntest_file = test[['Age', 'Gender', 'Annual Income', 'Marital Status','Number of Dependents', 'Education Level', 'Occupation', 'Health Score','Location', 'Policy Type', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'DayofPolicy', 'WeekofPolicy', 'MonthofPolicy', 'YearofPolicy']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:45.51824Z","iopub.execute_input":"2024-12-19T08:58:45.518643Z","iopub.status.idle":"2024-12-19T08:58:45.597225Z","shell.execute_reply.started":"2024-12-19T08:58:45.518611Z","shell.execute_reply":"2024-12-19T08:58:45.596027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgboost_preds = pmmodel_1.fit(X, y_log).predict(test_file)\nlgbm_preds = pmmodel_2.fit(X, y_log).predict(test_file)\nhgbr_preds = pmmodel_3.fit(X, y_log).predict(test_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:58:49.395705Z","iopub.execute_input":"2024-12-19T08:58:49.396087Z","iopub.status.idle":"2024-12-19T09:00:17.057041Z","shell.execute_reply.started":"2024-12-19T08:58:49.396059Z","shell.execute_reply":"2024-12-19T09:00:17.056044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"avg_preds = np.mean((xgboost_preds,lgbm_preds, hgbr_preds), axis=0)\nexp_preds = np.expm1(avg_preds)\nexp_preds.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:02:52.746879Z","iopub.execute_input":"2024-12-19T09:02:52.747264Z","iopub.status.idle":"2024-12-19T09:02:52.780425Z","shell.execute_reply.started":"2024-12-19T09:02:52.747238Z","shell.execute_reply":"2024-12-19T09:02:52.779377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission file copy\npm_sub_file = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\npm_sub_file.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:02:56.076151Z","iopub.execute_input":"2024-12-19T09:02:56.076545Z","iopub.status.idle":"2024-12-19T09:02:56.260162Z","shell.execute_reply.started":"2024-12-19T09:02:56.076507Z","shell.execute_reply":"2024-12-19T09:02:56.259236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pm_sub_final = pm_sub_file.copy()\npm_sub_final['Premium Amount'] = exp_preds\npm_sub_final.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:02:58.57305Z","iopub.execute_input":"2024-12-19T09:02:58.573382Z","iopub.status.idle":"2024-12-19T09:03:00.174145Z","shell.execute_reply.started":"2024-12-19T09:02:58.573358Z","shell.execute_reply":"2024-12-19T09:03:00.172998Z"}},"outputs":[],"execution_count":null}]}