{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:43:48.579341Z","iopub.execute_input":"2024-12-09T18:43:48.579838Z","iopub.status.idle":"2024-12-09T18:43:52.952611Z","shell.execute_reply.started":"2024-12-09T18:43:48.579788Z","shell.execute_reply":"2024-12-09T18:43:52.950782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:44:33.559807Z","iopub.execute_input":"2024-12-09T18:44:33.560301Z","iopub.status.idle":"2024-12-09T18:44:41.005443Z","shell.execute_reply.started":"2024-12-09T18:44:33.560258Z","shell.execute_reply":"2024-12-09T18:44:41.00441Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploring dataset and Feature Engineering","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:44:43.144542Z","iopub.execute_input":"2024-12-09T18:44:43.144977Z","iopub.status.idle":"2024-12-09T18:44:43.193064Z","shell.execute_reply.started":"2024-12-09T18:44:43.144936Z","shell.execute_reply":"2024-12-09T18:44:43.191852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:45:06.635233Z","iopub.execute_input":"2024-12-09T18:45:06.635636Z","iopub.status.idle":"2024-12-09T18:45:07.399021Z","shell.execute_reply.started":"2024-12-09T18:45:06.635603Z","shell.execute_reply":"2024-12-09T18:45:07.397573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:45:14.416137Z","iopub.execute_input":"2024-12-09T18:45:14.416548Z","iopub.status.idle":"2024-12-09T18:45:15.199108Z","shell.execute_reply.started":"2024-12-09T18:45:14.416516Z","shell.execute_reply":"2024-12-09T18:45:15.19788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:45:21.952212Z","iopub.execute_input":"2024-12-09T18:45:21.952653Z","iopub.status.idle":"2024-12-09T18:45:23.421794Z","shell.execute_reply.started":"2024-12-09T18:45:21.952616Z","shell.execute_reply":"2024-12-09T18:45:23.420251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:45:39.551327Z","iopub.execute_input":"2024-12-09T18:45:39.551723Z","iopub.status.idle":"2024-12-09T18:45:40.237953Z","shell.execute_reply.started":"2024-12-09T18:45:39.551686Z","shell.execute_reply":"2024-12-09T18:45:40.236475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values = train_df.isnull().sum()\n\nmissing_values = missing_values[missing_values > 0].sort_values(ascending=False)\n\nmissing_df = missing_values.reset_index()\nmissing_df.columns = ['Column', 'Missing Count']\n\nplt.figure(figsize=(12, 6))\nplt.bar(missing_df['Column'], missing_df['Missing Count'], color='skyblue')\nplt.title(\"Count of Missing Values per Column\", fontsize=16)\nplt.xlabel(\"Columns\", fontsize=12)\nplt.ylabel(\"Count of Missing Values\", fontsize=12)\nplt.xticks(rotation=45, ha='right', fontsize=10)\nplt.grid(axis='y', linestyle='--', alpha=0.7)\n\nfor i, value in enumerate(missing_df['Missing Count']):\n    plt.text(i, value + 0.02 * max(missing_df['Missing Count']), str(value), ha='center', fontsize=10)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:48:26.091224Z","iopub.execute_input":"2024-12-09T18:48:26.091618Z","iopub.status.idle":"2024-12-09T18:48:27.453676Z","shell.execute_reply.started":"2024-12-09T18:48:26.091584Z","shell.execute_reply":"2024-12-09T18:48:27.452458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.hist(train_df['Premium Amount'].dropna(), bins=50, color='skyblue', edgecolor='black', alpha=0.7)\nplt.title(\"Distribution of Premium Amount\", fontsize=16)\nplt.xlabel(\"Premium Amount\", fontsize=12)\nplt.ylabel(\"Frequency\", fontsize=12)\nplt.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Show the plot\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:49:17.600838Z","iopub.execute_input":"2024-12-09T18:49:17.601246Z","iopub.status.idle":"2024-12-09T18:49:17.974896Z","shell.execute_reply.started":"2024-12-09T18:49:17.60121Z","shell.execute_reply":"2024-12-09T18:49:17.973373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute Age\ntrain_df['Age'] = train_df['Age'].fillna(train_df['Age'].median())\n\n# Impute Marital Status\ntrain_df['Marital Status'] = train_df['Marital Status'].fillna(train_df['Marital Status'].mode()[0])\n\n# Impute Occupation\ntrain_df['Occupation'] = train_df['Occupation'].fillna('Unknown')\n\n# Impute Health Score\ntrain_df['Health Score'] = train_df['Health Score'].fillna(train_df['Health Score'].mean())\n\n# Impute Previous Claims\ntrain_df['Previous Claims'] = train_df['Previous Claims'].fillna(0)\n\n# Impute Credit Score\ntrain_df['Credit Score'] = train_df['Credit Score'].fillna(train_df['Credit Score'].median())\n\n# Impute Customer Feedback\ntrain_df['Customer Feedback'] = train_df['Customer Feedback'].fillna('No Feedback')\n\n# Create Age Groups\ntrain_df['Age Group'] = pd.cut(train_df['Age'], bins=[0, 25, 35, 45, 55, 65, 75, 85], \n                         labels=[\"0-25\", \"26-35\", \"36-45\", \"46-55\", \"56-65\", \"66-75\", \"76-85\"])\n\n# Impute Annual Income by Occupation and Age Group\ntrain_df['Annual Income'] = train_df.groupby(['Occupation', 'Age Group'])['Annual Income'].transform(\n    lambda x: x.fillna(x.median())\n)\ntrain_df['Annual Income'] = train_df['Annual Income'].fillna(train_df['Annual Income'].median())\n\n# Impute Number of Dependents by Marital Status and Age Group\ntrain_df['Number of Dependents'] = train_df.groupby(['Marital Status', 'Age Group'])['Number of Dependents'].transform(\n    lambda x: x.fillna(x.median())\n)\ntrain_df['Number of Dependents'] = train_df['Number of Dependents'].fillna(train_df['Number of Dependents'].median())\n\n# Handle null values for Insurance Duration using median\ntrain_df['Insurance Duration'] = train_df['Insurance Duration'].fillna(train_df['Insurance Duration'].median())\n\n# Handle null values for Vehicle Age using median\ntrain_df['Vehicle Age'] = train_df['Vehicle Age'].fillna(train_df['Vehicle Age'].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:49:49.22036Z","iopub.execute_input":"2024-12-09T18:49:49.220808Z","iopub.status.idle":"2024-12-09T18:49:50.64688Z","shell.execute_reply.started":"2024-12-09T18:49:49.220768Z","shell.execute_reply":"2024-12-09T18:49:50.645711Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- **Age:** Missing values were imputed with the median of the column, as age typically has a skewed distribution, and the median is robust to outliers.\n- **Marital Status:** Missing values were replaced with the mode (most frequent value) because marital status is categorical and mode represents the majority of the data.\n- **Occupation:** Missing values were replaced with the string 'Unknown', indicating that the occupation is not available.\n- **Health Score:** Missing values were filled with the mean, which works well for numerical values with a normal distribution.\n- **Previous Claims:** Missing values were replaced with 0, assuming that if claims are missing, there were likely no claims recorded.\n- **Credit Score:** Missing values were filled with the median because credit scores often have outliers, and the median is robust to extreme values.\n- **Customer Feedback:** Missing feedback values were replaced with 'No Feedback' to indicate no response or feedback from the customer.\n- **Age Group:** A new categorical column was created by binning the Age column into predefined ranges, such as 0-25, 26-35, etc. This allows easier analysis and segmentation based on age groups.\n- **Annual Income:** Missing values were filled using the median of grouped data (by Occupation and Age Group) to account for variations in income across different professions and age brackets. Remaining nulls were filled with the overall median.\n- **Number of Dependents:** Missing values were imputed using the median of grouped data (by Marital Status and Age Group) since dependents often correlate with these factors. Remaining nulls were filled with the overall median.\n- **Insurance Duration:** Missing values were replaced with the median as this column has a numerical, bounded range and typically follows a central tendency.\n- **Vehicle Age:** Missing values were filled with the median to handle the numerical nature of the data while avoiding distortion by outliers.","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:50:12.019843Z","iopub.execute_input":"2024-12-09T18:50:12.020356Z","iopub.status.idle":"2024-12-09T18:50:12.053278Z","shell.execute_reply.started":"2024-12-09T18:50:12.020312Z","shell.execute_reply":"2024-12-09T18:50:12.052055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:50:26.841699Z","iopub.execute_input":"2024-12-09T18:50:26.842236Z","iopub.status.idle":"2024-12-09T18:50:27.688617Z","shell.execute_reply.started":"2024-12-09T18:50:26.842182Z","shell.execute_reply":"2024-12-09T18:50:27.686411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"categorical_column = 'Gender'\n\ncategory_counts = train_df[categorical_column].value_counts()\n\nplt.figure(figsize=(8, 8))\nplt.pie(\n    category_counts,\n    labels=category_counts.index,\n    autopct='%1.1f%%',\n    startangle=90,\n    wedgeprops={'edgecolor': 'black'},\n    textprops={'fontsize': 12}\n)\nplt.title(f\"Proportion of {categorical_column}\", fontsize=16)\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:52:37.896034Z","iopub.execute_input":"2024-12-09T18:52:37.896462Z","iopub.status.idle":"2024-12-09T18:52:38.250558Z","shell.execute_reply.started":"2024-12-09T18:52:37.896424Z","shell.execute_reply":"2024-12-09T18:52:38.24923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def drawbarchart(column):\n    marital_status_counts = train_df[column].value_counts()\n\n    # Print the counts\n    print(marital_status_counts)\n\n    # Create the bar chart using Matplotlib\n    plt.figure(figsize=(10, 6))\n    plt.bar(marital_status_counts.index, marital_status_counts.values, color='skyblue', edgecolor='black', alpha=0.7)\n    plt.title(column+\" Distribution\", fontsize=16)\n    plt.xlabel(column, fontsize=12)\n    plt.ylabel(\"Count\", fontsize=12)\n    plt.xticks(rotation=45, fontsize=10, ha='right')\n    plt.grid(axis='y', linestyle='--', alpha=0.7)\n\n    # Show the plot\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:54:47.022601Z","iopub.execute_input":"2024-12-09T18:54:47.023023Z","iopub.status.idle":"2024-12-09T18:54:47.030616Z","shell.execute_reply.started":"2024-12-09T18:54:47.022984Z","shell.execute_reply":"2024-12-09T18:54:47.029249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drawbarchart('Marital Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:54:55.585408Z","iopub.execute_input":"2024-12-09T18:54:55.585866Z","iopub.status.idle":"2024-12-09T18:54:56.067187Z","shell.execute_reply.started":"2024-12-09T18:54:55.585828Z","shell.execute_reply":"2024-12-09T18:54:56.065912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drawbarchart('Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:55:08.58298Z","iopub.execute_input":"2024-12-09T18:55:08.583393Z","iopub.status.idle":"2024-12-09T18:55:09.044319Z","shell.execute_reply.started":"2024-12-09T18:55:08.583359Z","shell.execute_reply":"2024-12-09T18:55:09.043097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drawbarchart('Smoking Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:55:37.453918Z","iopub.execute_input":"2024-12-09T18:55:37.45442Z","iopub.status.idle":"2024-12-09T18:55:37.876462Z","shell.execute_reply.started":"2024-12-09T18:55:37.45438Z","shell.execute_reply":"2024-12-09T18:55:37.875269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = train_df.select_dtypes(include=['float64', 'int64']).columns\n\n# Compute the correlation matrix\ncorrelation_matrix = train_df[numerical_columns].corr()\n\n# Visualize the correlation matrix using a heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True)\nplt.title(\"Correlation Matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:56:06.894785Z","iopub.execute_input":"2024-12-09T18:56:06.895218Z","iopub.status.idle":"2024-12-09T18:56:08.128008Z","shell.execute_reply.started":"2024-12-09T18:56:06.895182Z","shell.execute_reply":"2024-12-09T18:56:08.126684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_feature = 'Annual Income'\ny_feature = 'Premium Amount'\n\nplt.figure(figsize=(10, 6))\nplt.scatter(\n    train_df[x_feature], \n    train_df[y_feature], \n    s=50,\n    alpha=0.6,\n    edgecolor='darkslategrey', \n    linewidth=1,\n    color='skyblue'\n)\n\nplt.title(f\"Scatter Plot of {x_feature} vs. {y_feature}\", fontsize=16)\nplt.xlabel(x_feature, fontsize=12)\nplt.ylabel(y_feature, fontsize=12)\n\nplt.grid(alpha=0.5, linestyle='--')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T18:57:43.419346Z","iopub.execute_input":"2024-12-09T18:57:43.419709Z","iopub.status.idle":"2024-12-09T18:57:48.13998Z","shell.execute_reply.started":"2024-12-09T18:57:43.419679Z","shell.execute_reply":"2024-12-09T18:57:48.13864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}