{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:47.248554Z","iopub.execute_input":"2024-12-17T17:19:47.249086Z","iopub.status.idle":"2024-12-17T17:19:47.256198Z","shell.execute_reply.started":"2024-12-17T17:19:47.249047Z","shell.execute_reply":"2024-12-17T17:19:47.254391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsample = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:47.258465Z","iopub.execute_input":"2024-12-17T17:19:47.258896Z","iopub.status.idle":"2024-12-17T17:19:56.61067Z","shell.execute_reply.started":"2024-12-17T17:19:47.258853Z","shell.execute_reply":"2024-12-17T17:19:56.609118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:56.612493Z","iopub.execute_input":"2024-12-17T17:19:56.613111Z","iopub.status.idle":"2024-12-17T17:19:57.318629Z","shell.execute_reply.started":"2024-12-17T17:19:56.613066Z","shell.execute_reply":"2024-12-17T17:19:57.317264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:57.321963Z","iopub.execute_input":"2024-12-17T17:19:57.32243Z","iopub.status.idle":"2024-12-17T17:19:57.992232Z","shell.execute_reply.started":"2024-12-17T17:19:57.322388Z","shell.execute_reply":"2024-12-17T17:19:57.990423Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#Preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:57.994174Z","iopub.execute_input":"2024-12-17T17:19:57.994612Z","iopub.status.idle":"2024-12-17T17:19:58.001348Z","shell.execute_reply.started":"2024-12-17T17:19:57.994574Z","shell.execute_reply":"2024-12-17T17:19:57.999113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def frequency_encode(my_df, drop_org=False):\n    df = my_df\n    df.columns = my_df.columns.str.replace(\" \", \"_\")\n    df = df.drop(\"id\", axis=\"columns\")\n    df[\"Policy_Start_Date\"] = pd.to_datetime(df[\"Policy_Start_Date\"])\n    df[\"Year\"] = df[\"Policy_Start_Date\"].dt.year.astype(\"category\")\n    df[\"Month\"] = df[\"Policy_Start_Date\"].dt.month.astype(\"category\")\n    df[\"Day\"] = df[\"Policy_Start_Date\"].dt.day.astype(\"category\")\n    df[\"Month_name\"] = df[\"Policy_Start_Date\"].dt.month_name()\n    df[\"Day_of_week\"] = df[\"Policy_Start_Date\"].dt.day_name()\n    df[\"Annual_Income_Health_Score_Ratio\"] = df[\"Health_Score\"] / df[\"Annual_Income\"]\n    df[\"Annual_Income_Age_Ratio\"] = df[\"Annual_Income\"] / df[\"Age\"]\n    df[\"Credit_Age\"] = df[\"Credit_Score\"] / df[\"Age\"]\n    df[\"Vehicle_Age_Insurance_Duration\"] = df[\"Vehicle_Age\"] / df[\"Insurance_Duration\"]\n    df[\"Policy_Start_Date\"] = df[\"Policy_Start_Date\"].dt.date\n\n    cat_columns = df.select_dtypes(include=['object','category']).columns\n    df_cols = df.columns.tolist()\n\n    new_cat_cols = []\n    for col in cat_columns:\n        traun_freq_encoding = df[col].value_counts().to_dict()\n        test_freq_encoding = df[col].value_counts().to_dict()\n\n        df[f\"{col}_freq\"] = df[col].map(traun_freq_encoding).astype('float32')\n\n        new_col_name = f\"{col}_freq\"\n        new_cat_cols.append(col)\n        df_cols.append(new_col_name)\n        if drop_org:\n            df_cols.remove(col)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:58.005497Z","iopub.execute_input":"2024-12-17T17:19:58.006448Z","iopub.status.idle":"2024-12-17T17:19:58.026504Z","shell.execute_reply.started":"2024-12-17T17:19:58.006379Z","shell.execute_reply":"2024-12-17T17:19:58.024647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train = frequency_encode(train.drop('Premium Amount', axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:19:58.027964Z","iopub.execute_input":"2024-12-17T17:19:58.028365Z","iopub.status.idle":"2024-12-17T17:20:05.171394Z","shell.execute_reply.started":"2024-12-17T17:19:58.028329Z","shell.execute_reply":"2024-12-17T17:20:05.169796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def one_hot_scal_encode(df):\n\n    cat_columns = df.select_dtypes(include=['object']).columns\n    int_columns = df.select_dtypes(include=['float64','float32', 'int64']).columns\n\n    df[int_columns] = df[int_columns].fillna(df[int_columns].median())\n    df[cat_columns] = df[cat_columns].fillna(\"Missing\")\n\n    df[cat_columns] = df[cat_columns].apply(lambda x: LabelEncoder().fit_transform(x))\n    df[int_columns] = StandardScaler().fit_transform(df[int_columns])\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:05.173302Z","iopub.execute_input":"2024-12-17T17:20:05.173729Z","iopub.status.idle":"2024-12-17T17:20:05.182541Z","shell.execute_reply.started":"2024-12-17T17:20:05.173692Z","shell.execute_reply":"2024-12-17T17:20:05.180681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train = one_hot_scal_encode(new_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:05.187379Z","iopub.execute_input":"2024-12-17T17:20:05.188346Z","iopub.status.idle":"2024-12-17T17:20:14.159452Z","shell.execute_reply.started":"2024-12-17T17:20:05.188284Z","shell.execute_reply":"2024-12-17T17:20:14.157824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:14.161554Z","iopub.execute_input":"2024-12-17T17:20:14.161978Z","iopub.status.idle":"2024-12-17T17:20:15.194053Z","shell.execute_reply.started":"2024-12-17T17:20:14.161941Z","shell.execute_reply":"2024-12-17T17:20:15.192533Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#Model","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom sklearn import metrics\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_squared_log_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:15.195675Z","iopub.execute_input":"2024-12-17T17:20:15.196118Z","iopub.status.idle":"2024-12-17T17:20:15.203513Z","shell.execute_reply.started":"2024-12-17T17:20:15.196079Z","shell.execute_reply":"2024-12-17T17:20:15.201833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = new_train\ny = np.log(train[\"Premium Amount\"])\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:15.205346Z","iopub.execute_input":"2024-12-17T17:20:15.205743Z","iopub.status.idle":"2024-12-17T17:20:16.886237Z","shell.execute_reply.started":"2024-12-17T17:20:15.205708Z","shell.execute_reply":"2024-12-17T17:20:16.884913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:16.88796Z","iopub.execute_input":"2024-12-17T17:20:16.888367Z","iopub.status.idle":"2024-12-17T17:20:16.895089Z","shell.execute_reply.started":"2024-12-17T17:20:16.888329Z","shell.execute_reply":"2024-12-17T17:20:16.893705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# params from optuna (look in comments)\nlgb_params = {\n'bagging_freq': 14, \n'min_data_in_leaf': 43, \n'max_depth': 16, \n'num_leaves': 161, \n'learning_rate': 0.013717061269682019, \n'feature_fraction': 0.8270498248315948, \n'bagging_fraction': 0.5944145124829179, \n'lambda_l1': 0.08871642518698214, \n'lambda_l2': 0.042853508674079666\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:16.897429Z","iopub.execute_input":"2024-12-17T17:20:16.89794Z","iopub.status.idle":"2024-12-17T17:20:16.913778Z","shell.execute_reply.started":"2024-12-17T17:20:16.897898Z","shell.execute_reply":"2024-12-17T17:20:16.911933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_final = LGBMRegressor(\n    **lgb_params,\n    objective= \"regression\",\n    metric= \"rmse\",\n    n_estimators= 500,\n    random_state=1,\n    verbose=-1\n)\nlgbm_final.fit(X_train, y_train)\ny_pred = np.exp(lgbm_final.predict(X_test))\nprint(\"RMSLE:\", rmsle(np.exp(y_test), y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:20:16.91603Z","iopub.execute_input":"2024-12-17T17:20:16.916611Z","iopub.status.idle":"2024-12-17T17:22:26.755982Z","shell.execute_reply.started":"2024-12-17T17:20:16.916553Z","shell.execute_reply":"2024-12-17T17:22:26.75452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#Prediction","metadata":{}},{"cell_type":"code","source":"new_test = one_hot_scal_encode(frequency_encode(test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:22:26.757437Z","iopub.execute_input":"2024-12-17T17:22:26.757793Z","iopub.status.idle":"2024-12-17T17:22:36.773798Z","shell.execute_reply.started":"2024-12-17T17:22:26.757758Z","shell.execute_reply":"2024-12-17T17:22:36.772221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_test = np.exp(lgbm_final.predict(new_test))\ny_pred_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:22:36.77891Z","iopub.execute_input":"2024-12-17T17:22:36.779453Z","iopub.status.idle":"2024-12-17T17:23:32.672459Z","shell.execute_reply.started":"2024-12-17T17:22:36.77941Z","shell.execute_reply":"2024-12-17T17:23:32.671032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample[\"Premium Amount\"] =  y_pred_test\n\nsample.to_csv('submission.csv',index=False)\nsample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:23:32.674707Z","iopub.execute_input":"2024-12-17T17:23:32.675329Z","iopub.status.idle":"2024-12-17T17:23:34.380128Z","shell.execute_reply.started":"2024-12-17T17:23:32.675258Z","shell.execute_reply":"2024-12-17T17:23:34.378888Z"}},"outputs":[],"execution_count":null}]}