{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:28.563158Z","iopub.execute_input":"2024-12-27T06:09:28.563457Z","iopub.status.idle":"2024-12-27T06:09:28.569375Z","shell.execute_reply.started":"2024-12-27T06:09:28.563433Z","shell.execute_reply":"2024-12-27T06:09:28.568601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsubmission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:28.617937Z","iopub.execute_input":"2024-12-27T06:09:28.618167Z","iopub.status.idle":"2024-12-27T06:09:34.441416Z","shell.execute_reply.started":"2024-12-27T06:09:28.618148Z","shell.execute_reply":"2024-12-27T06:09:34.440436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Shapes of the datasets","metadata":{}},{"cell_type":"code","source":"print(\"Training Data Shape\", train.shape)\nprint(\"Test Data Shape\", test.shape)\nprint(\"Submission Data Shape\", submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:34.442594Z","iopub.execute_input":"2024-12-27T06:09:34.442838Z","iopub.status.idle":"2024-12-27T06:09:34.44842Z","shell.execute_reply.started":"2024-12-27T06:09:34.442816Z","shell.execute_reply":"2024-12-27T06:09:34.44768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:34.449914Z","iopub.execute_input":"2024-12-27T06:09:34.450194Z","iopub.status.idle":"2024-12-27T06:09:34.99102Z","shell.execute_reply.started":"2024-12-27T06:09:34.450171Z","shell.execute_reply":"2024-12-27T06:09:34.990172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:34.992199Z","iopub.execute_input":"2024-12-27T06:09:34.992438Z","iopub.status.idle":"2024-12-27T06:09:35.063366Z","shell.execute_reply.started":"2024-12-27T06:09:34.992417Z","shell.execute_reply":"2024-12-27T06:09:35.06267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cleaned_data(df):\n    df['Age'] = df['Age'].fillna(df['Age'].mean())\n    df['Gender'] = df['Gender'].map({'Male':1, 'Female':0})\n    df['Annual Income'] = df['Annual Income'].fillna(df['Annual Income'].mean())\n    df['Marital Status'] = df['Marital Status'].map({'Single':1, 'Married':0, 'Divorced':2})\n    df['Marital Status'] = df['Marital Status'].fillna(1)\n    df['Number of Dependents'] = df['Number of Dependents'].fillna(df['Number of Dependents'].mean())\n    df['Education Level'] = df['Education Level'].map({\"Master's\":1, 'PhD':0, 'High School':2, \"Bachelor's\":3})\n    df['Occupation'] = df['Occupation'].map({'Employed':1, 'Self-Employed':0, 'Unemployed':2})\n    df['Occupation'] = df['Occupation'].fillna(1)\n    df['Health Score'] = df['Health Score'].fillna(df['Health Score'].mean())\n    df['Location'] = df['Location'].map({'Suburban':1, 'Rural':0, 'Urban':2})\n    df['Policy Type'] = df['Policy Type'].map({'Premium':1, 'Comprehensive':0, 'Basic':2})\n    df['Previous Claims'] = df['Previous Claims'].fillna(df['Previous Claims'].mean())\n    df['Vehicle Age'] = df['Vehicle Age'].fillna(df['Vehicle Age'].mean())\n    df['Credit Score'] = df['Credit Score'].fillna(df['Credit Score'].mean())\n    df['Insurance Duration'] = df['Insurance Duration'].fillna(df['Insurance Duration'].mean())\n    df.drop(columns=['Policy Start Date', 'id'], inplace=True)\n    df['Customer Feedback'] = df['Customer Feedback'].map({'Average':1, 'Poor':0, 'Good':2})\n    df['Customer Feedback'] = df['Customer Feedback'].fillna(1)\n    df['Smoking Status'] = df['Smoking Status'].map({'Yes':1, 'No':0})\n    df['Exercise Frequency'] = df['Exercise Frequency'].map({'Monthly':3, 'Weekly':2, 'Rarely':0, \"Daily\":1})\n    df['Property Type'] = df['Property Type'].map({'House':1, 'Apartment':0, 'Condo':2})\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:35.064016Z","iopub.execute_input":"2024-12-27T06:09:35.064251Z","iopub.status.idle":"2024-12-27T06:09:35.071761Z","shell.execute_reply.started":"2024-12-27T06:09:35.064227Z","shell.execute_reply":"2024-12-27T06:09:35.070918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_data(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:35.072433Z","iopub.execute_input":"2024-12-27T06:09:35.072625Z","iopub.status.idle":"2024-12-27T06:09:36.02932Z","shell.execute_reply.started":"2024-12-27T06:09:35.072608Z","shell.execute_reply":"2024-12-27T06:09:36.027805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cleaned_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:36.030206Z","iopub.execute_input":"2024-12-27T06:09:36.030526Z","iopub.status.idle":"2024-12-27T06:09:36.662609Z","shell.execute_reply.started":"2024-12-27T06:09:36.030489Z","shell.execute_reply":"2024-12-27T06:09:36.661757Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Our data is cleaned now, we can go ahead with modeling.","metadata":{}},{"cell_type":"markdown","source":"## Model training","metadata":{}},{"cell_type":"code","source":"y= train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:36.664998Z","iopub.execute_input":"2024-12-27T06:09:36.665245Z","iopub.status.idle":"2024-12-27T06:09:36.668658Z","shell.execute_reply.started":"2024-12-27T06:09:36.665223Z","shell.execute_reply":"2024-12-27T06:09:36.667751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = train.drop(columns=['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:36.669874Z","iopub.execute_input":"2024-12-27T06:09:36.670204Z","iopub.status.idle":"2024-12-27T06:09:36.758609Z","shell.execute_reply.started":"2024-12-27T06:09:36.670173Z","shell.execute_reply":"2024-12-27T06:09:36.757901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:36.759433Z","iopub.execute_input":"2024-12-27T06:09:36.759671Z","iopub.status.idle":"2024-12-27T06:09:36.763395Z","shell.execute_reply.started":"2024-12-27T06:09:36.759649Z","shell.execute_reply":"2024-12-27T06:09:36.762584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and test sets\n#X_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.2, random_state=42)\n#Training the data on complete training set","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:36.764271Z","iopub.execute_input":"2024-12-27T06:09:36.764616Z","iopub.status.idle":"2024-12-27T06:09:36.780578Z","shell.execute_reply.started":"2024-12-27T06:09:36.764578Z","shell.execute_reply":"2024-12-27T06:09:36.779906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize the RandomForestRegressor\nmodel =  LinearRegression()\n\n# Train the model\nmodel.fit(x, y)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:36.781298Z","iopub.execute_input":"2024-12-27T06:09:36.781566Z","iopub.status.idle":"2024-12-27T06:09:38.010545Z","shell.execute_reply.started":"2024-12-27T06:09:36.781535Z","shell.execute_reply":"2024-12-27T06:09:38.008453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predicting on the test dataset\ny_pred = model.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:38.011434Z","iopub.execute_input":"2024-12-27T06:09:38.011687Z","iopub.status.idle":"2024-12-27T06:09:38.066684Z","shell.execute_reply.started":"2024-12-27T06:09:38.011662Z","shell.execute_reply":"2024-12-27T06:09:38.064658Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluating the model with Root Mean Squared Logarithmic Error (RMSLE).","metadata":{}},{"cell_type":"code","source":"# Will evaluate the model on submission dataset\ny_test = submission['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:38.067645Z","iopub.execute_input":"2024-12-27T06:09:38.067905Z","iopub.status.idle":"2024-12-27T06:09:38.078396Z","shell.execute_reply.started":"2024-12-27T06:09:38.067879Z","shell.execute_reply":"2024-12-27T06:09:38.075631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_log_error\n\n# Ensure predictions and true values are non-negative\ny_pred = np.maximum(0, y_pred)  # Clip predictions to be non-negative\ny_test = np.maximum(0, y_test)  # Ensure actual values are non-negative\n\n# Calculate RMSLE\nrmsle = np.sqrt(mean_squared_log_error(y_test, y_pred))\nprint(f'Root Mean Squared Logarithmic Error (RMSLE): {rmsle}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:09:38.080256Z","iopub.execute_input":"2024-12-27T06:09:38.0812Z","iopub.status.idle":"2024-12-27T06:09:38.11682Z","shell.execute_reply.started":"2024-12-27T06:09:38.081159Z","shell.execute_reply":"2024-12-27T06:09:38.11584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Making submission File","metadata":{}},{"cell_type":"code","source":"mysub = pd.DataFrame(submission['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:11:43.71881Z","iopub.execute_input":"2024-12-27T06:11:43.719175Z","iopub.status.idle":"2024-12-27T06:11:43.723943Z","shell.execute_reply.started":"2024-12-27T06:11:43.71913Z","shell.execute_reply":"2024-12-27T06:11:43.723269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub['Premium Amount'] = y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:13:10.4437Z","iopub.execute_input":"2024-12-27T06:13:10.444015Z","iopub.status.idle":"2024-12-27T06:13:10.449541Z","shell.execute_reply.started":"2024-12-27T06:13:10.443988Z","shell.execute_reply":"2024-12-27T06:13:10.448779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:13:20.407948Z","iopub.execute_input":"2024-12-27T06:13:20.408242Z","iopub.status.idle":"2024-12-27T06:13:20.41724Z","shell.execute_reply.started":"2024-12-27T06:13:20.408218Z","shell.execute_reply":"2024-12-27T06:13:20.416267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mysub.to_csv('/kaggle/working/my_data.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T06:14:46.809412Z","iopub.execute_input":"2024-12-27T06:14:46.809697Z","iopub.status.idle":"2024-12-27T06:14:48.119879Z","shell.execute_reply.started":"2024-12-27T06:14:46.809675Z","shell.execute_reply":"2024-12-27T06:14:48.118832Z"}},"outputs":[],"execution_count":null}]}