{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:36.156418Z","iopub.execute_input":"2025-01-05T07:59:36.156777Z","iopub.status.idle":"2025-01-05T07:59:36.165091Z","shell.execute_reply.started":"2025-01-05T07:59:36.156746Z","shell.execute_reply":"2025-01-05T07:59:36.164014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:36.166311Z","iopub.execute_input":"2025-01-05T07:59:36.166595Z","iopub.status.idle":"2025-01-05T07:59:41.255275Z","shell.execute_reply.started":"2025-01-05T07:59:36.166548Z","shell.execute_reply":"2025-01-05T07:59:41.254247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:41.257249Z","iopub.execute_input":"2025-01-05T07:59:41.257553Z","iopub.status.idle":"2025-01-05T07:59:41.886012Z","shell.execute_reply.started":"2025-01-05T07:59:41.257527Z","shell.execute_reply":"2025-01-05T07:59:41.884993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:41.887387Z","iopub.execute_input":"2025-01-05T07:59:41.887651Z","iopub.status.idle":"2025-01-05T07:59:42.506916Z","shell.execute_reply.started":"2025-01-05T07:59:41.887628Z","shell.execute_reply":"2025-01-05T07:59:42.505795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dropping the columns with too much missing values\ntrain=data.drop(['Occupation','Previous Claims','Credit Score','Number of Dependents','id','Policy Start Date'],axis=1)\ncolumns_to_check = ['Vehicle Age', 'Insurance Duration']  # Specify the columns you want to check for missing values\ntrain = train.dropna(subset=columns_to_check)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:42.508048Z","iopub.execute_input":"2025-01-05T07:59:42.508431Z","iopub.status.idle":"2025-01-05T07:59:42.841431Z","shell.execute_reply.started":"2025-01-05T07:59:42.508392Z","shell.execute_reply":"2025-01-05T07:59:42.840508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#splitting the training and testing data\nX_train=train.drop('Premium Amount', axis=1)\ny_train=train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:42.842337Z","iopub.execute_input":"2025-01-05T07:59:42.842585Z","iopub.status.idle":"2025-01-05T07:59:42.970842Z","shell.execute_reply.started":"2025-01-05T07:59:42.842563Z","shell.execute_reply":"2025-01-05T07:59:42.969973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#filling in the missing data\nX_train['Gender'] = X_train['Gender'].fillna(X_train['Gender'].mode())\nX_train['Annual Income'] = X_train['Annual Income'].fillna(X_train['Annual Income'].median())\nX_train['Marital Status'] = X_train['Marital Status'].fillna(X_train['Marital Status'].mode()[0])\nX_train['Customer Feedback'] = X_train['Customer Feedback'].fillna('Not Available')\nX_train['Health Score'] = X_train['Health Score'].interpolate(method='linear')\nX_train['Age'] = X_train['Age'].fillna(X_train['Age'].median())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:42.973266Z","iopub.execute_input":"2025-01-05T07:59:42.973543Z","iopub.status.idle":"2025-01-05T07:59:43.554122Z","shell.execute_reply.started":"2025-01-05T07:59:42.97352Z","shell.execute_reply":"2025-01-05T07:59:43.553056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:43.556218Z","iopub.execute_input":"2025-01-05T07:59:43.556624Z","iopub.status.idle":"2025-01-05T07:59:43.573824Z","shell.execute_reply.started":"2025-01-05T07:59:43.556582Z","shell.execute_reply":"2025-01-05T07:59:43.572987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nsns.countplot(data=X_train, x='Policy Type', order=X_train['Policy Type'].value_counts().index, hue=X_train['Smoking Status'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:43.574992Z","iopub.execute_input":"2025-01-05T07:59:43.575363Z","iopub.status.idle":"2025-01-05T07:59:45.022598Z","shell.execute_reply.started":"2025-01-05T07:59:43.575321Z","shell.execute_reply":"2025-01-05T07:59:45.021506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=X_train, x='Smoking Status', order=X_train['Smoking Status'].value_counts().index, hue=X_train['Gender'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:45.023521Z","iopub.execute_input":"2025-01-05T07:59:45.023778Z","iopub.status.idle":"2025-01-05T07:59:46.539052Z","shell.execute_reply.started":"2025-01-05T07:59:45.023756Z","shell.execute_reply":"2025-01-05T07:59:46.537995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=X_train, x='Property Type', order=X_train['Property Type'].value_counts().index)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:46.540173Z","iopub.execute_input":"2025-01-05T07:59:46.540559Z","iopub.status.idle":"2025-01-05T07:59:47.355042Z","shell.execute_reply.started":"2025-01-05T07:59:46.540529Z","shell.execute_reply":"2025-01-05T07:59:47.353782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subset=X_train[['Age','Health Score','Vehicle Age','Insurance Duration']]\nsubset=subset.sample(n=5000, random_state=42)\nsns.pairplot(subset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:47.356144Z","iopub.execute_input":"2025-01-05T07:59:47.356514Z","iopub.status.idle":"2025-01-05T07:59:52.113432Z","shell.execute_reply.started":"2025-01-05T07:59:47.356474Z","shell.execute_reply":"2025-01-05T07:59:52.112081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test=pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:52.114465Z","iopub.execute_input":"2025-01-05T07:59:52.114759Z","iopub.status.idle":"2025-01-05T07:59:55.186174Z","shell.execute_reply.started":"2025-01-05T07:59:52.114733Z","shell.execute_reply":"2025-01-05T07:59:55.185029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, MinMaxScaler, OrdinalEncoder \nfrom sklearn.pipeline import Pipeline\n\neducation_order = sorted(X_train['Education Level'].dropna().unique())  # Sort unique values\npolicy_order = sorted(X_train['Policy Type'].dropna().unique())         # Sort unique values\nfeedback_order = sorted(X_train['Customer Feedback'].dropna().unique())\nexercise_order = sorted(X_train['Exercise Frequency'].dropna().unique())\n\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')  # For nominal columns\nordinal_transformer = OrdinalEncoder(categories=[education_order, policy_order, feedback_order, exercise_order])  # For ordinal columns\nnumerical_transformer = StandardScaler()  # For scaling numerical column\npreprocessor = ColumnTransformer(\n    transformers=[\n        # Apply OneHotEncoder to nominal columns\n        ('nominal', nominal_transformer, ['Gender', 'Marital Status', 'Location', 'Smoking Status', 'Property Type']),\n        \n        # Apply OrdinalEncoder to ordinal columns\n        ('ordinal', ordinal_transformer, ['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']),\n        \n        # Apply StandardScaler to Annual Income\n        ('numerical', numerical_transformer, ['Annual Income']),\n    ],\n    remainder='passthrough'  # Keeps any other columns untouched (optional)\n)\n\n# Create a pipeline with the preprocessor\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor)\n])\n\n# Fit and transform the dataset\nX_train_transformed = pipeline.fit_transform(X_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:55.187744Z","iopub.execute_input":"2025-01-05T07:59:55.188157Z","iopub.status.idle":"2025-01-05T07:59:58.928842Z","shell.execute_reply.started":"2025-01-05T07:59:55.188116Z","shell.execute_reply":"2025-01-05T07:59:58.927972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nregr = RandomForestRegressor(max_depth=5, random_state=42,oob_score=True)\nregr.fit(X_train_transformed, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T07:59:58.92974Z","iopub.execute_input":"2025-01-05T07:59:58.930237Z","iopub.status.idle":"2025-01-05T08:07:02.015642Z","shell.execute_reply.started":"2025-01-05T07:59:58.930149Z","shell.execute_reply":"2025-01-05T08:07:02.014621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test=X_test.drop(['Occupation','Previous Claims','Credit Score','Number of Dependents','id','Policy Start Date'],axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T08:07:02.016551Z","iopub.execute_input":"2025-01-05T08:07:02.016806Z","iopub.status.idle":"2025-01-05T08:07:02.139403Z","shell.execute_reply.started":"2025-01-05T08:07:02.016784Z","shell.execute_reply":"2025-01-05T08:07:02.138014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX_test['Annual Income'] = X_test['Annual Income'].fillna(X_test['Annual Income'].median())\nX_test['Marital Status'] = X_test['Marital Status'].fillna(X_test['Marital Status'].mode()[0])\nX_test['Customer Feedback'] = X_test['Customer Feedback'].fillna('Not Available')\nX_test['Health Score'] = X_test['Health Score'].interpolate(method='linear')\nX_test['Age'] = X_test['Age'].fillna(X_test['Age'].median())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T08:09:59.704765Z","iopub.execute_input":"2025-01-05T08:09:59.705105Z","iopub.status.idle":"2025-01-05T08:09:59.904355Z","shell.execute_reply.started":"2025-01-05T08:09:59.705079Z","shell.execute_reply":"2025-01-05T08:09:59.903444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.dropna(inplace=True)\nX_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T08:10:34.345384Z","iopub.execute_input":"2025-01-05T08:10:34.345712Z","iopub.status.idle":"2025-01-05T08:10:35.169493Z","shell.execute_reply.started":"2025-01-05T08:10:34.345686Z","shell.execute_reply":"2025-01-05T08:10:35.168419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"education_order = sorted(X_test['Education Level'].dropna().unique())  # Sort unique values\npolicy_order = sorted(X_test['Policy Type'].dropna().unique())         # Sort unique values\nfeedback_order = sorted(X_test['Customer Feedback'].dropna().unique())\nexercise_order = sorted(X_test['Exercise Frequency'].dropna().unique())\n\nnominal_transformer = OneHotEncoder(handle_unknown='ignore')  # For nominal columns\nordinal_transformer = OrdinalEncoder(categories=[education_order, policy_order, feedback_order, exercise_order])  # For ordinal columns\nnumerical_transformer = StandardScaler()  # For scaling numerical column\npreprocessor = ColumnTransformer(\n    transformers=[\n        # Apply OneHotEncoder to nominal columns\n        ('nominal', nominal_transformer, ['Gender', 'Marital Status', 'Location', 'Smoking Status', 'Property Type']),\n        \n        # Apply OrdinalEncoder to ordinal columns\n        ('ordinal', ordinal_transformer, ['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']),\n        \n        # Apply StandardScaler to Annual Income\n        ('numerical', numerical_transformer, ['Annual Income']),\n    ],\n    remainder='passthrough'  # Keeps any other columns untouched (optional)\n)\n\n# Create a pipeline with the preprocessor\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor)\n])\n\n# Fit and transform the dataset\nX_test_transformed = pipeline.fit_transform(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T08:10:38.361828Z","iopub.execute_input":"2025-01-05T08:10:38.362209Z","iopub.status.idle":"2025-01-05T08:10:40.635014Z","shell.execute_reply.started":"2025-01-05T08:10:38.36218Z","shell.execute_reply":"2025-01-05T08:10:40.634154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, r2_score\n\n# Access the OOB Score\noob_score = regr.oob_score_\nprint(f'Out-of-Bag Score: {oob_score}')\n\n# Making predictions on the same data or new data\npredictions = regr.predict(X_train_transformed)\n\n# Evaluating the model\nmse = mean_squared_error(y_train, predictions)\nprint(f'Mean Squared Error: {mse}')\n\nr2 = r2_score(y_train, predictions)\nprint(f'R-squared: {r2}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T08:16:17.939187Z","iopub.execute_input":"2025-01-05T08:16:17.93954Z","iopub.status.idle":"2025-01-05T08:16:22.509035Z","shell.execute_reply.started":"2025-01-05T08:16:17.939511Z","shell.execute_reply":"2025-01-05T08:16:22.507957Z"}},"outputs":[],"execution_count":null}]}