{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:58:10.958767Z","iopub.execute_input":"2024-12-20T09:58:10.95934Z","iopub.status.idle":"2024-12-20T09:58:10.970946Z","shell.execute_reply.started":"2024-12-20T09:58:10.959309Z","shell.execute_reply":"2024-12-20T09:58:10.969487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndata_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:01.452237Z","iopub.execute_input":"2024-12-20T09:59:01.45277Z","iopub.status.idle":"2024-12-20T09:59:09.942898Z","shell.execute_reply.started":"2024-12-20T09:59:01.452734Z","shell.execute_reply":"2024-12-20T09:59:09.941276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Preprocessing</h1>","metadata":{}},{"cell_type":"markdown","source":"The preprocessing part is consisting of:\n* Extracting the *Policy Start Date*'s day, month, and year.\n* Turning the categorical columns into categorical and fill the missing values as 'None'.","metadata":{}},{"cell_type":"code","source":"def cat_maker(data, cat_cols):\n    for i in cat_cols:\n        data.loc[:, i] = data[i].fillna('None')\n        data[i] = data[i].astype('category')\n    return data\n\ndef datetime_feature(data):\n    date_time = pd.to_datetime(data['Policy Start Date'])\n    data['year'] = date_time.dt.year\n    data['month'] = date_time.dt.month\n    data['day'] = date_time.dt.day\n    data['month_sin'] = np.sin(2 * np.pi * data['month'] / 12)\n    data['month_cos'] = np.cos(2 * np.pi * data['month'] / 12)\n    data['day_sin'] = np.sin(2 * np.pi * data['day'] / 31)\n    data['day_cos'] = np.cos(2 * np.pi * data['day'] / 31)\n    #data['year'] = data['year'].astype('category')\n    data.drop(columns=['Policy Start Date', 'month', 'day'], axis=1, inplace=True)\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:12.167294Z","iopub.execute_input":"2024-12-20T09:59:12.167664Z","iopub.status.idle":"2024-12-20T09:59:12.175406Z","shell.execute_reply.started":"2024-12-20T09:59:12.16762Z","shell.execute_reply":"2024-12-20T09:59:12.174254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation',\n       'Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'Customer Feedback', 'year']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:14.534773Z","iopub.execute_input":"2024-12-20T09:59:14.535212Z","iopub.status.idle":"2024-12-20T09:59:14.540353Z","shell.execute_reply.started":"2024-12-20T09:59:14.535179Z","shell.execute_reply":"2024-12-20T09:59:14.539088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train = datetime_feature(data_train)\ndata_train = cat_maker(data_train, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:16.275232Z","iopub.execute_input":"2024-12-20T09:59:16.275715Z","iopub.status.idle":"2024-12-20T09:59:18.82189Z","shell.execute_reply.started":"2024-12-20T09:59:16.275665Z","shell.execute_reply":"2024-12-20T09:59:18.820727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Feature selection</h1>","metadata":{}},{"cell_type":"markdown","source":"<p>The idea is to find features that has a high correlation towards the target value. However, categorical columns can't be used for correlation. An alternative is to use ANOVA test (since all of the categorical columns has a few ordinal values).</p>","metadata":{}},{"cell_type":"code","source":"from scipy.stats import f_oneway\n\ndef anova_test(data, cat_cols, target):\n    anova_res = []\n    for col in cat_cols:\n        groups = [data[target][data[col] == cat] for cat in data[col].unique()]\n        f_stat, p_val = f_oneway(*groups)\n        anova_res.append([col, f_stat, p_val])\n    return anova_res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:23.115394Z","iopub.execute_input":"2024-12-20T09:59:23.115776Z","iopub.status.idle":"2024-12-20T09:59:23.121898Z","shell.execute_reply.started":"2024-12-20T09:59:23.115748Z","shell.execute_reply":"2024-12-20T09:59:23.120388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"anova_res = anova_test(data_train, cat_cols, 'Premium Amount')\nanova_df = pd.DataFrame(anova_res, columns = ['Column', 'F Statistic', 'P-value'])\nanova_df = anova_df.sort_values(ascending = True, by = 'P-value')\nanova_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:26.046937Z","iopub.execute_input":"2024-12-20T09:59:26.047306Z","iopub.status.idle":"2024-12-20T09:59:26.919782Z","shell.execute_reply.started":"2024-12-20T09:59:26.047279Z","shell.execute_reply":"2024-12-20T09:59:26.918583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train_num = data_train.drop(columns=cat_cols)\ndata_train_num.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:29.243279Z","iopub.execute_input":"2024-12-20T09:59:29.243727Z","iopub.status.idle":"2024-12-20T09:59:29.311066Z","shell.execute_reply.started":"2024-12-20T09:59:29.243693Z","shell.execute_reply":"2024-12-20T09:59:29.30997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_col = data_train_num.corr()['Premium Amount'].drop(['id', 'Premium Amount'])\ncorr_df = corr_col.reset_index()\ncorr_df.columns = ['Columns', 'Correlation']\ncorr_df = corr_df.sort_values(ascending = False, by = 'Correlation')\ncorr_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:31.273737Z","iopub.execute_input":"2024-12-20T09:59:31.274127Z","iopub.status.idle":"2024-12-20T09:59:32.066951Z","shell.execute_reply.started":"2024-12-20T09:59:31.274098Z","shell.execute_reply":"2024-12-20T09:59:32.065612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p>Now, we take every categorical column that has P-value less than 0.05 (this is a common interval meaning that this column is significant towards the target variable), and correlations that is higher than 0.01 or less than -0.01</p>","metadata":{}},{"cell_type":"code","source":"#selected columns\nsel_cat = np.array(anova_df[anova_df['P-value'] < 0.05]['Column'])\nsel_num = np.array(corr_df[(corr_df['Correlation'] > 0.01) | (corr_df['Correlation'] < -0.01)]['Columns'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:34.413384Z","iopub.execute_input":"2024-12-20T09:59:34.413757Z","iopub.status.idle":"2024-12-20T09:59:34.421291Z","shell.execute_reply.started":"2024-12-20T09:59:34.413728Z","shell.execute_reply":"2024-12-20T09:59:34.419976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sel_cols = np.append(sel_num, sel_cat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:36.630835Z","iopub.execute_input":"2024-12-20T09:59:36.631239Z","iopub.status.idle":"2024-12-20T09:59:36.636195Z","shell.execute_reply.started":"2024-12-20T09:59:36.631206Z","shell.execute_reply":"2024-12-20T09:59:36.634923Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Modelling</h1>","metadata":{}},{"cell_type":"markdown","source":"<p>The use for filling all the missing values is unecessary in this case since some of them actually means something. Read this useful discussion <a href='https://www.kaggle.com/competitions/playground-series-s4e12/discussion/552165'>here</a>.</p>","metadata":{}},{"cell_type":"code","source":"import optuna\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold, train_test_split\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:39.495Z","iopub.execute_input":"2024-12-20T09:59:39.495341Z","iopub.status.idle":"2024-12-20T09:59:39.502049Z","shell.execute_reply.started":"2024-12-20T09:59:39.495314Z","shell.execute_reply":"2024-12-20T09:59:39.500515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = data_train['Premium Amount']\nX = data_train[sel_cols]\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:41.06539Z","iopub.execute_input":"2024-12-20T09:59:41.06604Z","iopub.status.idle":"2024-12-20T09:59:41.242603Z","shell.execute_reply.started":"2024-12-20T09:59:41.066003Z","shell.execute_reply":"2024-12-20T09:59:41.241458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\n\n'''\nkf = KFold(n_splits = 5, shuffle = True, random_state = 1)\ndef objective(trial):\n    param = {\n        'objective': 'regression',  # Regression problem\n        'metric': 'rmse',  # RMSE as the evaluation metric\n        'boosting_type': 'gbdt',\n        'num_leaves': trial.suggest_int('num_leaves', 20, 100),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 1000),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n        'feature_fraction': trial.suggest_float('feature_fraction', 0.6, 1.0),\n        'max_depth': trial.suggest_int('max_depth', -1, 20),  # -1 means no limit\n        'lambda_l1': trial.suggest_float('lambda_l1', 0.0, 5.0),\n        'lambda_l2': trial.suggest_float('lambda_l2', 0.0, 5.0),\n        \"verbosity\": -1,\n    }\n    cv = []\n    for train_index, valid_index in kf.split(X_train):\n        X_train_cv, X_valid = X_train.iloc[train_index], X_train.iloc[valid_index]\n        y_train_cv, y_valid = np.log1p(y_train).iloc[train_index], np.log1p(y_train).iloc[valid_index]\n        model = lgb.LGBMRegressor(**param, random_state=1)\n        model.fit(X_train_cv, y_train_cv)\n        preds = model.predict(X_valid)\n        error = np.sqrt(mean_squared_error(y_valid, preds))\n        cv.append(error)\n    return np.mean(cv)\n\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=50)\n\nprint(\"Best Parameters:\", study.best_params)\nprint(\"Best RMSLE:\", study.best_value)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:25:56.262142Z","iopub.execute_input":"2024-12-20T09:25:56.262603Z","iopub.status.idle":"2024-12-20T09:25:56.270187Z","shell.execute_reply.started":"2024-12-20T09:25:56.262568Z","shell.execute_reply":"2024-12-20T09:25:56.268776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Best Parameters: {'num_leaves': 43, 'n_estimators': 726, 'learning_rate': 0.02163170133738646, 'feature_fraction': 0.9838318541664448, 'max_depth': 0, 'lambda_l1': 3.0412536629291314, 'lambda_l2': 1.7276192279459714}\n#Best RMSLE: 1.0460590754803685","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_param = {'num_leaves': 43, 'n_estimators': 726, 'learning_rate': 0.02163170133738646, 'feature_fraction': 0.9838318541664448, 'max_depth': 0, 'lambda_l1': 3.0412536629291314, 'lambda_l2': 1.7276192279459714}\nmodel = lgb.LGBMRegressor(**best_param, random_state=1, verbosity=-1)\nmodel.fit(X_train, np.log1p(y_train))\npreds = model.predict(X_test)\nerror = np.sqrt(mean_squared_error(np.log1p(y_test), preds))\nprint(f\"LGBM's RMSLE: {error}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:59:44.770306Z","iopub.execute_input":"2024-12-20T09:59:44.770774Z","iopub.status.idle":"2024-12-20T10:00:14.028485Z","shell.execute_reply.started":"2024-12-20T09:59:44.770739Z","shell.execute_reply":"2024-12-20T10:00:14.026903Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Final train</h1>","metadata":{}},{"cell_type":"code","source":"model = lgb.LGBMRegressor(**best_param, random_state=1, verbosity=-1)\nmodel.fit(X, np.log1p(y))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:01:56.06088Z","iopub.execute_input":"2024-12-20T10:01:56.061627Z","iopub.status.idle":"2024-12-20T10:02:23.423605Z","shell.execute_reply.started":"2024-12-20T10:01:56.061543Z","shell.execute_reply":"2024-12-20T10:02:23.422353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#transform test data\ndata_test = datetime_feature(data_test)\ndata_test = cat_maker(data_test, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:02:30.782462Z","iopub.execute_input":"2024-12-20T10:02:30.782952Z","iopub.status.idle":"2024-12-20T10:02:32.506424Z","shell.execute_reply.started":"2024-12-20T10:02:30.782925Z","shell.execute_reply":"2024-12-20T10:02:32.505159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = model.predict(data_test[sel_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:02:34.346698Z","iopub.execute_input":"2024-12-20T10:02:34.347087Z","iopub.status.idle":"2024-12-20T10:02:58.180277Z","shell.execute_reply.started":"2024-12-20T10:02:34.347058Z","shell.execute_reply":"2024-12-20T10:02:58.178982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_res = pd.DataFrame({'id':data_test['id'], 'Premium Amount':np.expm1(preds)})\ndf_res.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:03:05.660008Z","iopub.execute_input":"2024-12-20T10:03:05.660382Z","iopub.status.idle":"2024-12-20T10:03:07.345434Z","shell.execute_reply.started":"2024-12-20T10:03:05.660352Z","shell.execute_reply":"2024-12-20T10:03:07.343953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_res.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:03:08.744164Z","iopub.execute_input":"2024-12-20T10:03:08.744542Z","iopub.status.idle":"2024-12-20T10:03:08.755018Z","shell.execute_reply.started":"2024-12-20T10:03:08.744512Z","shell.execute_reply":"2024-12-20T10:03:08.753825Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p>If you find this helpful, please leave an upvote!</p><p>Any sugestions are also welcome.</p>","metadata":{}}]}