{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:57:45.982599Z","iopub.execute_input":"2024-12-01T10:57:45.983369Z","iopub.status.idle":"2024-12-01T10:57:46.414018Z","shell.execute_reply.started":"2024-12-01T10:57:45.983307Z","shell.execute_reply":"2024-12-01T10:57:46.412938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import GridSearchCV","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:57:47.447964Z","iopub.execute_input":"2024-12-01T10:57:47.448522Z","iopub.status.idle":"2024-12-01T10:57:47.454957Z","shell.execute_reply.started":"2024-12-01T10:57:47.448485Z","shell.execute_reply":"2024-12-01T10:57:47.453647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading the dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:57:47.837104Z","iopub.execute_input":"2024-12-01T10:57:47.83763Z","iopub.status.idle":"2024-12-01T10:57:59.067Z","shell.execute_reply.started":"2024-12-01T10:57:47.837566Z","shell.execute_reply":"2024-12-01T10:57:59.065997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:57:59.069158Z","iopub.execute_input":"2024-12-01T10:57:59.069716Z","iopub.status.idle":"2024-12-01T10:57:59.119855Z","shell.execute_reply.started":"2024-12-01T10:57:59.069661Z","shell.execute_reply":"2024-12-01T10:57:59.118331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:57:59.121299Z","iopub.execute_input":"2024-12-01T10:57:59.121718Z","iopub.status.idle":"2024-12-01T10:57:59.814282Z","shell.execute_reply.started":"2024-12-01T10:57:59.121682Z","shell.execute_reply":"2024-12-01T10:57:59.813086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.drop(columns = ['Policy Start Date', 'id'], axis = 1)\ntest_df = test_df.drop(columns = ['Policy Start Date'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:57:59.8165Z","iopub.execute_input":"2024-12-01T10:57:59.816839Z","iopub.status.idle":"2024-12-01T10:58:00.098101Z","shell.execute_reply.started":"2024-12-01T10:57:59.816804Z","shell.execute_reply":"2024-12-01T10:58:00.097014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:00.099401Z","iopub.execute_input":"2024-12-01T10:58:00.099777Z","iopub.status.idle":"2024-12-01T10:58:00.1072Z","shell.execute_reply.started":"2024-12-01T10:58:00.099736Z","shell.execute_reply":"2024-12-01T10:58:00.105997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:00.108567Z","iopub.execute_input":"2024-12-01T10:58:00.108867Z","iopub.status.idle":"2024-12-01T10:58:00.696188Z","shell.execute_reply.started":"2024-12-01T10:58:00.108837Z","shell.execute_reply":"2024-12-01T10:58:00.695115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Impute missing values for numerical columns","metadata":{}},{"cell_type":"code","source":"numerical_cols = train_df.select_dtypes(include=['float64']).columns\nimputer = SimpleImputer(strategy='median')\ntrain_df[numerical_cols] = imputer.fit_transform(train_df[numerical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:00.69772Z","iopub.execute_input":"2024-12-01T10:58:00.698062Z","iopub.status.idle":"2024-12-01T10:58:02.845955Z","shell.execute_reply.started":"2024-12-01T10:58:00.698029Z","shell.execute_reply":"2024-12-01T10:58:02.844875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute missing values for categorical columns","metadata":{}},{"cell_type":"code","source":"categorical_cols = train_df.select_dtypes(include=['object']).columns\nfor col in categorical_cols:\n    train_df[col] = train_df[col].fillna(train_df[col].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:02.847489Z","iopub.execute_input":"2024-12-01T10:58:02.847842Z","iopub.status.idle":"2024-12-01T10:58:04.66893Z","shell.execute_reply.started":"2024-12-01T10:58:02.847806Z","shell.execute_reply":"2024-12-01T10:58:04.667805Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding categorical variables","metadata":{}},{"cell_type":"code","source":"label_encoders = {}\nfor col in categorical_cols:\n    le = LabelEncoder()\n    train_df[col] = le.fit_transform(train_df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:04.67019Z","iopub.execute_input":"2024-12-01T10:58:04.67051Z","iopub.status.idle":"2024-12-01T10:58:07.129696Z","shell.execute_reply.started":"2024-12-01T10:58:04.670477Z","shell.execute_reply":"2024-12-01T10:58:07.128622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Split into features (X) and target (y)","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['Premium Amount'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:07.133775Z","iopub.execute_input":"2024-12-01T10:58:07.134227Z","iopub.status.idle":"2024-12-01T10:58:07.259029Z","shell.execute_reply.started":"2024-12-01T10:58:07.13419Z","shell.execute_reply":"2024-12-01T10:58:07.257635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:07.26097Z","iopub.execute_input":"2024-12-01T10:58:07.261412Z","iopub.status.idle":"2024-12-01T10:58:07.29435Z","shell.execute_reply.started":"2024-12-01T10:58:07.261359Z","shell.execute_reply":"2024-12-01T10:58:07.292949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y_log, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:07.295879Z","iopub.execute_input":"2024-12-01T10:58:07.296324Z","iopub.status.idle":"2024-12-01T10:58:07.852728Z","shell.execute_reply.started":"2024-12-01T10:58:07.296284Z","shell.execute_reply":"2024-12-01T10:58:07.851535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  RandomForestRegressor model","metadata":{}},{"cell_type":"code","source":"rf_model = RandomForestRegressor(n_estimators=500, max_depth=10, min_samples_split=5, random_state=42, n_jobs=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:07.854059Z","iopub.execute_input":"2024-12-01T10:58:07.854437Z","iopub.status.idle":"2024-12-01T10:58:07.859757Z","shell.execute_reply.started":"2024-12-01T10:58:07.854402Z","shell.execute_reply":"2024-12-01T10:58:07.858465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:58:07.860864Z","iopub.execute_input":"2024-12-01T10:58:07.86126Z","iopub.status.idle":"2024-12-01T11:16:45.676449Z","shell.execute_reply.started":"2024-12-01T10:58:07.861226Z","shell.execute_reply":"2024-12-01T11:16:45.675307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_log = rf_model.predict(X_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:16:45.677983Z","iopub.execute_input":"2024-12-01T11:16:45.67847Z","iopub.status.idle":"2024-12-01T11:16:49.103256Z","shell.execute_reply.started":"2024-12-01T11:16:45.678421Z","shell.execute_reply":"2024-12-01T11:16:49.101859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = np.expm1(y_pred_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:16:49.104871Z","iopub.execute_input":"2024-12-01T11:16:49.105254Z","iopub.status.idle":"2024-12-01T11:16:49.116017Z","shell.execute_reply.started":"2024-12-01T11:16:49.105218Z","shell.execute_reply":"2024-12-01T11:16:49.114776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_val = np.expm1(y_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:16:49.117355Z","iopub.execute_input":"2024-12-01T11:16:49.117664Z","iopub.status.idle":"2024-12-01T11:16:49.132493Z","shell.execute_reply.started":"2024-12-01T11:16:49.117633Z","shell.execute_reply":"2024-12-01T11:16:49.131096Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Calculate RMSLE","metadata":{}},{"cell_type":"code","source":"rmsle_rf = np.sqrt(mean_squared_log_error(y_val, y_pred))\nprint(f'Random Forest RMSLE: {rmsle_rf}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:16:49.1358Z","iopub.execute_input":"2024-12-01T11:16:49.136323Z","iopub.status.idle":"2024-12-01T11:16:49.155401Z","shell.execute_reply.started":"2024-12-01T11:16:49.13628Z","shell.execute_reply":"2024-12-01T11:16:49.15414Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Data Preprocessing","metadata":{}},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:12.954373Z","iopub.execute_input":"2024-12-01T11:27:12.954803Z","iopub.status.idle":"2024-12-01T11:27:12.962213Z","shell.execute_reply.started":"2024-12-01T11:27:12.954764Z","shell.execute_reply":"2024-12-01T11:27:12.961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:13.184711Z","iopub.execute_input":"2024-12-01T11:27:13.185679Z","iopub.status.idle":"2024-12-01T11:27:13.581896Z","shell.execute_reply.started":"2024-12-01T11:27:13.18563Z","shell.execute_reply":"2024-12-01T11:27:13.580799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:13.583774Z","iopub.execute_input":"2024-12-01T11:27:13.58413Z","iopub.status.idle":"2024-12-01T11:27:13.978202Z","shell.execute_reply.started":"2024-12-01T11:27:13.584062Z","shell.execute_reply":"2024-12-01T11:27:13.976924Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute missing values for numerical columns","metadata":{}},{"cell_type":"code","source":"numerical_cols = test_df.select_dtypes(include=['float64']).columns\nimputer = SimpleImputer(strategy='median')\ntest_df[numerical_cols] = imputer.fit_transform(test_df[numerical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:13.980301Z","iopub.execute_input":"2024-12-01T11:27:13.980656Z","iopub.status.idle":"2024-12-01T11:27:15.218515Z","shell.execute_reply.started":"2024-12-01T11:27:13.98062Z","shell.execute_reply":"2024-12-01T11:27:15.217339Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute missing values for categorical columns","metadata":{}},{"cell_type":"code","source":"categorical_cols = test_df.select_dtypes(include=['object']).columns\nfor col in categorical_cols:\n    test_df[col] = test_df[col].fillna(test_df[col].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:15.220331Z","iopub.execute_input":"2024-12-01T11:27:15.220743Z","iopub.status.idle":"2024-12-01T11:27:16.44991Z","shell.execute_reply.started":"2024-12-01T11:27:15.220704Z","shell.execute_reply":"2024-12-01T11:27:16.44835Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding categorical variables","metadata":{}},{"cell_type":"code","source":"label_encoders = {}\nfor col in categorical_cols:\n    le = LabelEncoder()\n    test_df[col] = le.fit_transform(test_df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:16.451767Z","iopub.execute_input":"2024-12-01T11:27:16.452256Z","iopub.status.idle":"2024-12-01T11:27:18.040122Z","shell.execute_reply.started":"2024-12-01T11:27:16.452198Z","shell.execute_reply":"2024-12-01T11:27:18.038912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = test_df.drop(columns = ['id'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:18.042465Z","iopub.execute_input":"2024-12-01T11:27:18.04284Z","iopub.status.idle":"2024-12-01T11:27:18.133545Z","shell.execute_reply.started":"2024-12-01T11:27:18.042803Z","shell.execute_reply":"2024-12-01T11:27:18.132607Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Tuning HyperParameter","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparam_grid = {\n    'n_estimators': [100, 500, 1000],\n    'max_depth': [None, 10, 20, 30],\n    'min_samples_split': [2, 5, 10]\n}\n\n# grid_search = GridSearchCV(estimator=RandomForestRegressor(random_state=42, n_jobs=-1), param_grid=param_grid, cv=3, scoring='neg_root_mean_squared_error')\n# grid_search.fit(X_train, y_train)\n\n# print(\"Best parameters:\", grid_search.best_params_)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:18.134702Z","iopub.execute_input":"2024-12-01T11:27:18.134993Z","iopub.status.idle":"2024-12-01T11:27:18.140693Z","shell.execute_reply.started":"2024-12-01T11:27:18.134964Z","shell.execute_reply":"2024-12-01T11:27:18.139433Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test predictions","metadata":{}},{"cell_type":"code","source":"test_pred = rf_model.predict(test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:18.141999Z","iopub.execute_input":"2024-12-01T11:27:18.142369Z","iopub.status.idle":"2024-12-01T11:27:30.701824Z","shell.execute_reply.started":"2024-12-01T11:27:18.142335Z","shell.execute_reply":"2024-12-01T11:27:30.700662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:30.703263Z","iopub.execute_input":"2024-12-01T11:27:30.70361Z","iopub.status.idle":"2024-12-01T11:27:30.711361Z","shell.execute_reply.started":"2024-12-01T11:27:30.703576Z","shell.execute_reply":"2024-12-01T11:27:30.710089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = np.expm1(test_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:30.712684Z","iopub.execute_input":"2024-12-01T11:27:30.713254Z","iopub.status.idle":"2024-12-01T11:27:30.745632Z","shell.execute_reply.started":"2024-12-01T11:27:30.713151Z","shell.execute_reply":"2024-12-01T11:27:30.74388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:30.747058Z","iopub.execute_input":"2024-12-01T11:27:30.747526Z","iopub.status.idle":"2024-12-01T11:27:30.754522Z","shell.execute_reply.started":"2024-12-01T11:27:30.747479Z","shell.execute_reply":"2024-12-01T11:27:30.753384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': test_df['id'],\n    'Premium Amount': test_predictions\n})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(\"Prediction file has been created\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:27:30.756913Z","iopub.execute_input":"2024-12-01T11:27:30.757266Z","iopub.status.idle":"2024-12-01T11:27:32.413259Z","shell.execute_reply.started":"2024-12-01T11:27:30.757232Z","shell.execute_reply":"2024-12-01T11:27:32.412173Z"}},"outputs":[],"execution_count":null}]}