{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np \nimport pandas as pd \n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\npd.set_option('display.float_format', lambda x: '%.5f' % x)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:22.573677Z","iopub.execute_input":"2024-12-13T01:12:22.574861Z","iopub.status.idle":"2024-12-13T01:12:26.123544Z","shell.execute_reply.started":"2024-12-13T01:12:22.574803Z","shell.execute_reply":"2024-12-13T01:12:26.12209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsample = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:26.125968Z","iopub.execute_input":"2024-12-13T01:12:26.126559Z","iopub.status.idle":"2024-12-13T01:12:37.600324Z","shell.execute_reply.started":"2024-12-13T01:12:26.126519Z","shell.execute_reply":"2024-12-13T01:12:37.599185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA\n- Take an initial look at our data","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:37.601636Z","iopub.execute_input":"2024-12-13T01:12:37.60213Z","iopub.status.idle":"2024-12-13T01:12:37.641113Z","shell.execute_reply.started":"2024-12-13T01:12:37.602078Z","shell.execute_reply":"2024-12-13T01:12:37.639949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:37.644272Z","iopub.execute_input":"2024-12-13T01:12:37.645327Z","iopub.status.idle":"2024-12-13T01:12:38.328839Z","shell.execute_reply.started":"2024-12-13T01:12:37.645271Z","shell.execute_reply":"2024-12-13T01:12:38.327619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:38.330213Z","iopub.execute_input":"2024-12-13T01:12:38.330556Z","iopub.status.idle":"2024-12-13T01:12:39.071026Z","shell.execute_reply.started":"2024-12-13T01:12:38.330524Z","shell.execute_reply":"2024-12-13T01:12:39.069823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:39.072126Z","iopub.execute_input":"2024-12-13T01:12:39.072486Z","iopub.status.idle":"2024-12-13T01:12:39.093549Z","shell.execute_reply.started":"2024-12-13T01:12:39.072455Z","shell.execute_reply":"2024-12-13T01:12:39.092141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:39.095185Z","iopub.execute_input":"2024-12-13T01:12:39.095572Z","iopub.status.idle":"2024-12-13T01:12:39.551101Z","shell.execute_reply.started":"2024-12-13T01:12:39.095533Z","shell.execute_reply":"2024-12-13T01:12:39.549757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:39.552297Z","iopub.execute_input":"2024-12-13T01:12:39.552751Z","iopub.status.idle":"2024-12-13T01:12:39.99936Z","shell.execute_reply.started":"2024-12-13T01:12:39.552699Z","shell.execute_reply":"2024-12-13T01:12:39.998168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:54:52.247278Z","iopub.execute_input":"2024-12-13T01:54:52.247736Z","iopub.status.idle":"2024-12-13T01:54:52.262406Z","shell.execute_reply.started":"2024-12-13T01:54:52.247687Z","shell.execute_reply":"2024-12-13T01:54:52.261156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for duplicates - none\nduplicates = train_df.duplicated()\ndisplay(train_df[duplicates])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:40.000877Z","iopub.execute_input":"2024-12-13T01:12:40.001251Z","iopub.status.idle":"2024-12-13T01:12:41.849243Z","shell.execute_reply.started":"2024-12-13T01:12:40.001211Z","shell.execute_reply":"2024-12-13T01:12:41.848146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:41.853813Z","iopub.execute_input":"2024-12-13T01:12:41.854181Z","iopub.status.idle":"2024-12-13T01:12:42.494402Z","shell.execute_reply.started":"2024-12-13T01:12:41.854147Z","shell.execute_reply":"2024-12-13T01:12:42.493125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:42.4957Z","iopub.execute_input":"2024-12-13T01:12:42.496152Z","iopub.status.idle":"2024-12-13T01:12:42.92968Z","shell.execute_reply.started":"2024-12-13T01:12:42.496107Z","shell.execute_reply":"2024-12-13T01:12:42.928431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualizations\n- Start with categorical columns","metadata":{}},{"cell_type":"code","source":"cat_cols = train_df.select_dtypes(include=['object', 'category']).columns.tolist()\ncat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:42.930864Z","iopub.execute_input":"2024-12-13T01:12:42.931213Z","iopub.status.idle":"2024-12-13T01:12:43.112776Z","shell.execute_reply.started":"2024-12-13T01:12:42.931179Z","shell.execute_reply":"2024-12-13T01:12:43.111213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"values = train_df['Gender'].value_counts()\nplt.figure(figsize=(8, 8))\nplt.pie(values, labels=[f'{index}\\n({value:,} samples, {value/len(train_df)*100:.1f}%)' \n                      for index, value in values.items()], autopct='%1.1f%%')\nplt.title('Gender Distribution')\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:43.114494Z","iopub.execute_input":"2024-12-13T01:12:43.114832Z","iopub.status.idle":"2024-12-13T01:12:43.429517Z","shell.execute_reply.started":"2024-12-13T01:12:43.114799Z","shell.execute_reply":"2024-12-13T01:12:43.427923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(12, 12))\nplt.tight_layout(pad=3.0)\n\ncolumns = ['Marital Status', 'Occupation', 'Location','Policy Type']\n\nfor ax, col in zip(axes.ravel(), columns):\n   values = train_df[col].value_counts()\n   ax.pie(values, labels=[f'{index}\\n({value:,}, {value/len(train_df)*100:.1f}%)' \n                         for index, value in values.items()], autopct='%1.1f%%')\n   ax.set_title(f'{col} Distribution')\n\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:43.431378Z","iopub.execute_input":"2024-12-13T01:12:43.43214Z","iopub.status.idle":"2024-12-13T01:12:44.489103Z","shell.execute_reply.started":"2024-12-13T01:12:43.432024Z","shell.execute_reply":"2024-12-13T01:12:44.487584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(12, 12))\nplt.tight_layout(pad=3.0)\n\ncolumns = ['Customer Feedback', 'Smoking Status','Exercise Frequency','Property Type']\n\nfor ax, col in zip(axes.ravel(), columns):\n   values = train_df[col].value_counts(dropna=False)\n   ax.pie(values, labels=[f'{index}\\n({value:,}, {value/len(train_df)*100:.1f}%)' \n                         for index, value in values.items()], autopct='%1.1f%%')\n   ax.set_title(f'{col} Distribution')\n\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:57:29.26515Z","iopub.execute_input":"2024-12-13T01:57:29.26571Z","iopub.status.idle":"2024-12-13T01:57:30.334378Z","shell.execute_reply.started":"2024-12-13T01:57:29.26565Z","shell.execute_reply":"2024-12-13T01:57:30.333124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Convert the 'Policy Start Date' to datetime and create two new features. We can get more detailed info from the datetime later if we want.","metadata":{}},{"cell_type":"code","source":"# Convert to datetime and create Policy Age in years\ntrain_df['Policy Start Date'] = pd.to_datetime(train_df['Policy Start Date'])\ncurrent_date = pd.Timestamp('2024-12-12')  # Using current date from conversation\ntrain_df['Policy Age'] = (current_date - train_df['Policy Start Date']).dt.total_seconds() / (365.25 * 24 * 60 * 60)\n\n# Calculate remaining time\ntrain_df['Policy Time Remaining'] = train_df['Insurance Duration'] - train_df['Policy Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:45.538271Z","iopub.execute_input":"2024-12-13T01:12:45.538731Z","iopub.status.idle":"2024-12-13T01:12:46.006788Z","shell.execute_reply.started":"2024-12-13T01:12:45.538678Z","shell.execute_reply":"2024-12-13T01:12:46.005506Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Now take a look at the numeric columns","metadata":{}},{"cell_type":"code","source":"num_cols = train_df.select_dtypes(include=['int64', 'float64']).columns.tolist()\nnum_cols.remove('id')\n\nnum_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:31:24.055392Z","iopub.execute_input":"2024-12-13T01:31:24.056098Z","iopub.status.idle":"2024-12-13T01:31:24.209361Z","shell.execute_reply.started":"2024-12-13T01:31:24.056038Z","shell.execute_reply":"2024-12-13T01:31:24.208015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate skewness for numerical features\nskewness = train_df[num_cols].skew().sort_values(ascending=False)\nprint(\"Skewness of numerical features:\")\nprint(skewness)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:02:28.027108Z","iopub.execute_input":"2024-12-13T03:02:28.027551Z","iopub.status.idle":"2024-12-13T03:02:28.255224Z","shell.execute_reply.started":"2024-12-13T03:02:28.027515Z","shell.execute_reply":"2024-12-13T03:02:28.254148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Outliers for numerical columns\nfor col in num_cols:\n   Q1 = train_df[col].quantile(0.25)\n   Q3 = train_df[col].quantile(0.75)\n   IQR = Q3 - Q1\n   outliers = train_df[(train_df[col] < (Q1 - 1.5 * IQR)) | (train_df[col] > (Q3 + 1.5 * IQR))]\n   print(f\"\\n{col}:\")\n   print(f\"Number of outliers: {len(outliers)}\")\n   print(f\"Percentage of outliers: {(len(outliers)/len(train_df))*100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:02:37.52307Z","iopub.execute_input":"2024-12-13T03:02:37.523495Z","iopub.status.idle":"2024-12-13T03:02:38.378947Z","shell.execute_reply.started":"2024-12-13T03:02:37.523457Z","shell.execute_reply":"2024-12-13T03:02:38.377667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(data=train_df, x='Age', bins=47, color='skyblue', edgecolor='black')\nplt.title('Distribution of Customer Age', pad=15, fontsize=14)\nplt.xlabel('Age (years)', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.grid(True, alpha=0.3)\nplt.ylim(20000, 30000)\nsns.despine(left=True, bottom=True)\n\navg = train_df['Age'].mean()\nmed = train_df['Age'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:46.157131Z","iopub.execute_input":"2024-12-13T01:12:46.157527Z","iopub.status.idle":"2024-12-13T01:12:47.19888Z","shell.execute_reply.started":"2024-12-13T01:12:46.157484Z","shell.execute_reply":"2024-12-13T01:12:47.197788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train_df, x='Annual Income', bins=30, color='green', edgecolor='black')\nplt.title('Distribution of Customer Annual Income', pad=15, fontsize=14)\nplt.xlabel('Income', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.grid(True, alpha=0.3)\nsns.despine(left=True, bottom=True)\n\n# Add summary statistics\navg = train_df['Annual Income'].mean()\nmed = train_df['Annual Income'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:47.200496Z","iopub.execute_input":"2024-12-13T01:12:47.200952Z","iopub.status.idle":"2024-12-13T01:12:48.00582Z","shell.execute_reply.started":"2024-12-13T01:12:47.200897Z","shell.execute_reply":"2024-12-13T01:12:48.004547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(12, 12))\nplt.tight_layout(pad=3.0)\n\ncolumns = ['Number of Dependents','Insurance Duration']\n\nfor ax, col in zip(axes.ravel(), columns):\n   values = train_df[col].value_counts()\n   ax.pie(values, labels=[f'{index}\\n({value:,}, {value/len(train_df)*100:.1f}%)' \n                         for index, value in values.items()], autopct='%1.1f%%')\n   ax.set_title(f'{col} Distribution')\n\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T02:00:22.929579Z","iopub.execute_input":"2024-12-13T02:00:22.930214Z","iopub.status.idle":"2024-12-13T02:00:23.442963Z","shell.execute_reply.started":"2024-12-13T02:00:22.930166Z","shell.execute_reply":"2024-12-13T02:00:23.441297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.barplot(x=values.index, y=values.values, color='skyblue')\nplt.title('Distribution of Previous Claims', fontsize=14, pad=15)\nplt.xlabel('Number of Claims', fontsize=12)\nplt.ylabel('Number of Customers', fontsize=12)\nplt.ylim(125000, 142000)  # Increased upper limit to make room\nfor i, v in enumerate(values.values):\n  plt.text(i, 128000, f'{v:,}\\n({v/len(train_df)*100:.1f}%)', \n           ha='center', va='bottom')\nplt.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T02:03:30.906246Z","iopub.execute_input":"2024-12-13T02:03:30.906654Z","iopub.status.idle":"2024-12-13T02:03:31.308355Z","shell.execute_reply.started":"2024-12-13T02:03:30.906617Z","shell.execute_reply":"2024-12-13T02:03:31.307117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train_df, x='Health Score', bins=100, color='purple', edgecolor='black')\nplt.title('Distribution of Health Scores', pad=15, fontsize=14)\nplt.xlabel('Score', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.grid(True, alpha=0.3)\nsns.despine(left=True, bottom=True)\n\n# Add summary statistics\navg = train_df['Health Score'].mean()\nmed = train_df['Health Score'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:48.983472Z","iopub.execute_input":"2024-12-13T01:12:48.983968Z","iopub.status.idle":"2024-12-13T01:12:49.89534Z","shell.execute_reply.started":"2024-12-13T01:12:48.983916Z","shell.execute_reply":"2024-12-13T01:12:49.893906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train_df, x='Vehicle Age', bins=17, color='green', edgecolor='black')\nplt.title('Distribution of Vehicle Ages', pad=15, fontsize=14)\nplt.xlabel('Age', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.grid(True, alpha=0.3)\nplt.ylim(40000, 130000)\nsns.despine(left=True, bottom=True)\n\n# Add summary statistics\navg = train_df['Vehicle Age'].mean()\nmed = train_df['Vehicle Age'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:49.897035Z","iopub.execute_input":"2024-12-13T01:12:49.89755Z","iopub.status.idle":"2024-12-13T01:12:50.707341Z","shell.execute_reply.started":"2024-12-13T01:12:49.8975Z","shell.execute_reply":"2024-12-13T01:12:50.706108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train_df, x='Premium Amount', bins=100, color='blue', edgecolor='black')\nplt.title('Distribution of Premium Amounts', pad=15, fontsize=14)\nplt.xlabel('Premiums', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.grid(True, alpha=0.3)\nsns.despine(left=True, bottom=True)\n\n# Add summary statistics\navg = train_df['Premium Amount'].mean()\nmed = train_df['Premium Amount'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:50.708973Z","iopub.execute_input":"2024-12-13T01:12:50.709483Z","iopub.status.idle":"2024-12-13T01:12:51.532138Z","shell.execute_reply.started":"2024-12-13T01:12:50.709415Z","shell.execute_reply":"2024-12-13T01:12:51.530678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.kdeplot(data=train_df, x='Policy Age', color='red', linewidth=2)\nplt.title('Policy Age Distribution', pad=15, fontsize=14)\nplt.xlabel('Age (Years)', fontsize=12)\nplt.ylabel('Density', fontsize=12)\nplt.grid(True, alpha=0.3)\nsns.despine(left=True, bottom=True)\n\navg = train_df['Policy Age'].mean()\nmed = train_df['Policy Age'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:51.533724Z","iopub.execute_input":"2024-12-13T01:12:51.535166Z","iopub.status.idle":"2024-12-13T01:12:56.718807Z","shell.execute_reply.started":"2024-12-13T01:12:51.535123Z","shell.execute_reply":"2024-12-13T01:12:56.717525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.kdeplot(data=train_df, x='Policy Time Remaining', color='red', linewidth=2)\nplt.title('Policy Time Remaining Distribution', pad=15, fontsize=14)\nplt.xlabel('Time Left (Years)', fontsize=12)\nplt.ylabel('Density', fontsize=12)\nplt.grid(True, alpha=0.3)\nsns.despine(left=True, bottom=True)\n\navg = train_df['Policy Time Remaining'].mean()\nmed = train_df['Policy Time Remaining'].median()\nplt.axvline(avg, color='red', linestyle='--', alpha=0.8, label=f'Mean: {avg:.1f}')\nplt.axvline(med, color='green', linestyle='--', alpha=0.8, label=f'Median: {med:.1f}')\nplt.legend()\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:12:56.720311Z","iopub.execute_input":"2024-12-13T01:12:56.720672Z","iopub.status.idle":"2024-12-13T01:13:02.01774Z","shell.execute_reply.started":"2024-12-13T01:12:56.720629Z","shell.execute_reply":"2024-12-13T01:13:02.016281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate correlations\ncorr_matrix = train_df[num_cols].corr()\n\n# Create heatmap\nplt.figure(figsize=(12, 10))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', center=0, fmt='.2f')\nplt.title('Correlation Heatmap of Numeric Features')\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:31:30.48251Z","iopub.execute_input":"2024-12-13T01:31:30.482909Z","iopub.status.idle":"2024-12-13T01:31:31.805023Z","shell.execute_reply.started":"2024-12-13T01:31:30.482865Z","shell.execute_reply":"2024-12-13T01:31:31.803407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(data=train_df, x='Premium Amount', bins=50)\nplt.title('Distribution of Premium Amount')\nplt.xlabel('Premium Amount')\nplt.ylabel('Count')\nplt.show()\n\n# Boxplot to show outliers\nplt.figure(figsize=(10, 2))\nplt.boxplot(train_df['Premium Amount'], vert=False)\nplt.title('Premium Amount Boxplot')\nplt.xlabel('Premium Amount')\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:53:28.88344Z","iopub.execute_input":"2024-12-13T01:53:28.883858Z","iopub.status.idle":"2024-12-13T01:53:29.752549Z","shell.execute_reply.started":"2024-12-13T01:53:28.883819Z","shell.execute_reply":"2024-12-13T01:53:29.751152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Continued Visualizations\n- Data is looking very balanced.\n- Next examine combined features and target feature","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nfeedback_policy = pd.crosstab(train_df['Customer Feedback'], \n                             train_df['Policy Type'], normalize='columns') * 100\nsns.heatmap(feedback_policy, annot=True, fmt='.1f', cmap='RdYlGn')\nplt.title('Customer Feedback Distribution by Policy Type (%)')\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:13:35.290585Z","iopub.execute_input":"2024-12-13T01:13:35.290888Z","iopub.status.idle":"2024-12-13T01:13:35.89947Z","shell.execute_reply.started":"2024-12-13T01:13:35.290857Z","shell.execute_reply":"2024-12-13T01:13:35.898148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\nsns.barplot(data=train_df, x='Number of Dependents', y='Premium Amount',\n            hue='Marital Status', errorbar=None)\nplt.legend(bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.title('Average Premium by Dependents and Marital Status')\nplt.ylim(1050, 1150)\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:18:32.037582Z","iopub.execute_input":"2024-12-13T01:18:32.03798Z","iopub.status.idle":"2024-12-13T01:18:32.999355Z","shell.execute_reply.started":"2024-12-13T01:18:32.037945Z","shell.execute_reply":"2024-12-13T01:18:32.998118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.scatterplot(data=train_df.sample(10000), \n                x='Health Score', \n                y='Credit Score',\n                hue='Previous Claims',\n                size='Premium Amount',\n                sizes=(20, 200),\n                alpha=0.6)\nplt.title('Health Score vs Credit Score\\nColored by Previous Claims, Size by Premium')\nplt.legend(bbox_to_anchor=(1.05, 1))\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T02:05:36.447416Z","iopub.execute_input":"2024-12-13T02:05:36.447839Z","iopub.status.idle":"2024-12-13T02:05:37.654249Z","shell.execute_reply.started":"2024-12-13T02:05:36.447804Z","shell.execute_reply":"2024-12-13T02:05:37.653121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\nlocation_health = train_df.groupby(['Location', 'Exercise Frequency'])['Health Score'].mean().unstack()\nsns.heatmap(location_health, annot=True, fmt='.1f', cmap='YlOrRd')\nplt.title('Average Health Score by Location and Exercise Frequency')\nplt.tight_layout()\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:22:09.303902Z","iopub.execute_input":"2024-12-13T01:22:09.304938Z","iopub.status.idle":"2024-12-13T01:22:09.760139Z","shell.execute_reply.started":"2024-12-13T01:22:09.304878Z","shell.execute_reply":"2024-12-13T01:22:09.758802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average Premium Amount vs Health Score bins\ntrain_df['Health_Score_Bins'] = pd.cut(train_df['Health Score'], bins=10)\nhealth_premium = train_df.groupby('Health_Score_Bins')['Premium Amount'].mean().reset_index()\n\nplt.figure(figsize=(10, 6))\nplt.plot(range(len(health_premium)), health_premium['Premium Amount'], marker='o')\nplt.xticks(range(len(health_premium)), health_premium['Health_Score_Bins'], rotation=45)\nplt.xlabel('Health Score Range')\nplt.ylabel('Average Premium Amount')\nplt.title('Average Premium Amount vs Health Score Ranges')\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:43:40.68812Z","iopub.execute_input":"2024-12-13T01:43:40.688561Z","iopub.status.idle":"2024-12-13T01:43:41.073939Z","shell.execute_reply.started":"2024-12-13T01:43:40.688523Z","shell.execute_reply":"2024-12-13T01:43:41.072512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average Premium by Location and Property Type\nlocation_property_premium = train_df.groupby(['Location', 'Property Type'])['Premium Amount'].mean().unstack()\nplt.figure(figsize=(12, 8))\nsns.heatmap(location_property_premium, annot=True, fmt='.0f', cmap='YlOrRd')\nplt.title('Average Premium by Location and Property Type')\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:51:42.16059Z","iopub.execute_input":"2024-12-13T01:51:42.160991Z","iopub.status.idle":"2024-12-13T01:51:42.766216Z","shell.execute_reply.started":"2024-12-13T01:51:42.160947Z","shell.execute_reply":"2024-12-13T01:51:42.764555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average Premium over Vehicle Age\nvehicle_premium = train_df.groupby('Vehicle Age')['Premium Amount'].mean().reset_index()\nplt.figure(figsize=(10, 6))\nplt.plot(vehicle_premium['Vehicle Age'], vehicle_premium['Premium Amount'])\nplt.title('Average Premium vs Vehicle Age')\nplt.xlabel('Vehicle Age')\nplt.ylabel('Premium Amount')\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:51:13.058731Z","iopub.execute_input":"2024-12-13T01:51:13.0592Z","iopub.status.idle":"2024-12-13T01:51:13.385459Z","shell.execute_reply.started":"2024-12-13T01:51:13.05916Z","shell.execute_reply":"2024-12-13T01:51:13.383918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Premium Amount by Marital Status\nplt.figure(figsize=(10, 6))\nsns.violinplot(data=train_df, x='Marital Status', y='Premium Amount')\nplt.title('Premium Distribution by Marital Status')\nplt.xticks(rotation=45)\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:50:47.961259Z","iopub.execute_input":"2024-12-13T01:50:47.961712Z","iopub.status.idle":"2024-12-13T01:50:51.453401Z","shell.execute_reply.started":"2024-12-13T01:50:47.961675Z","shell.execute_reply":"2024-12-13T01:50:51.45228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average Premium by Policy Type and Smoking Status\navg_premium = train_df.groupby(['Policy Type', 'Smoking Status'])['Premium Amount'].mean().unstack()\navg_premium.plot(kind='bar', figsize=(10, 6))\nplt.title('Average Premium by Policy Type and Smoking Status')\nplt.xlabel('Policy Type')\nplt.ylabel('Average Premium Amount')\nplt.ylim(1050, 1150)\nplt.legend(title='Smoking Status')\nplt.xticks(rotation=45)\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:49:30.301086Z","iopub.execute_input":"2024-12-13T01:49:30.301508Z","iopub.status.idle":"2024-12-13T01:49:30.948197Z","shell.execute_reply.started":"2024-12-13T01:49:30.301468Z","shell.execute_reply":"2024-12-13T01:49:30.946766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average Premium by Exercise Frequency\ntrain_df.groupby('Exercise Frequency')['Premium Amount'].mean().plot(kind='bar', figsize=(10, 6))\nplt.title('Average Premium by Exercise Frequency')\nplt.xlabel('Exercise Frequency')\nplt.ylabel('Average Premium Amount')\nplt.xticks(rotation=45)\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T01:49:47.178809Z","iopub.execute_input":"2024-12-13T01:49:47.179232Z","iopub.status.idle":"2024-12-13T01:49:47.549307Z","shell.execute_reply.started":"2024-12-13T01:49:47.179194Z","shell.execute_reply":"2024-12-13T01:49:47.548162Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Mutual Information Analysis","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import mutual_info_regression\n\n\nX = train_df[num_cols].drop('Premium Amount', axis=1)\n\n# Encode categorical columns\nfor col in cat_cols:\n    # Label encode each category\n    X[col] = pd.Categorical(train_df[col]).codes\n    \ny = train_df['Premium Amount']\n\n# Calculate mutual information\nmi_scores = mutual_info_regression(X.fillna(0), y)\n\n# Create dataframe of scores\nmi_df = pd.DataFrame({'Feature': X.columns, 'MI Score': mi_scores})\nmi_df = mi_df.sort_values('MI Score', ascending=False)\n\n# Plot\nplt.figure(figsize=(10, 6))\nsns.barplot(data=mi_df, x='MI Score', y='Feature')\nplt.title('Mutual Information Scores with Premium Amount')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T02:19:01.51095Z","iopub.execute_input":"2024-12-13T02:19:01.51137Z","iopub.status.idle":"2024-12-13T02:28:13.518735Z","shell.execute_reply.started":"2024-12-13T02:19:01.511331Z","shell.execute_reply":"2024-12-13T02:28:13.517483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# VIF\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\n# Select numerical columns and fill NA\nX = train_df[num_cols].fillna(0)\n\n# Calculate VIF for each feature\nvif_data = pd.DataFrame()\nvif_data[\"Feature\"] = X.columns\nvif_data[\"VIF\"] = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\nprint(\"\\nVIF Scores:\")\nprint(vif_data.sort_values('VIF', ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:07:27.350385Z","iopub.execute_input":"2024-12-13T03:07:27.350833Z","iopub.status.idle":"2024-12-13T03:07:46.026592Z","shell.execute_reply.started":"2024-12-13T03:07:27.350796Z","shell.execute_reply":"2024-12-13T03:07:46.02341Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LGBM ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import mean_squared_error, r2_score\nimport lightgbm as lgb\n\ndef rmsle_metric(y_pred, dtrain):\n    y_true = dtrain.get_label()\n    # Add 1 to avoid log(0)\n    y_true = np.clip(y_true, a_min=0, a_max=None) + 1\n    y_pred = np.clip(y_pred, a_min=0, a_max=None) + 1\n    return 'RMSLE', np.sqrt(mean_squared_error(np.log(y_true), np.log(y_pred))), False\n\n# Drop specified columns and separate features/target\nX = train_df.drop(['id', 'Policy Start Date', 'Premium Amount'], axis=1)\ny = train_df['Premium Amount']\n\n# Identify numeric and categorical columns\nnumeric_cols = X.select_dtypes(include=['int64', 'float64']).columns\ncategorical_cols = X.select_dtypes(include=['object']).columns\n\n# One-hot encode categorical variables\nencoder = OneHotEncoder(drop='first',sparse_output=False, handle_unknown='ignore')\nencoded_cats = encoder.fit_transform(X[categorical_cols])\nencoded_feature_names = encoder.get_feature_names_out(categorical_cols)\n\n# Combine numeric and encoded categorical data\nX_numeric = X[numeric_cols].values\nX_processed = np.hstack([X_numeric, encoded_cats])\nfeature_names = list(numeric_cols) + list(encoded_feature_names)\n\n# Initialize KFold and storage for predictions\nkf = KFold(n_splits=3, shuffle=True, random_state=42)\noof_predictions = np.zeros(len(X))\nfeature_importance_df = pd.DataFrame()\n\n# Training and cross-validation\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_processed)):\n    X_train, X_val = X_processed[train_idx], X_processed[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    train_data = lgb.Dataset(X_train, label=y_train)\n    val_data = lgb.Dataset(X_val, label=y_val)\n    \n    params = {\n    'objective': 'regression',\n    'metric': 'custom',\n    'boosting_type': 'gbdt',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n    'force_row_wise': True,\n    'verbose' : 1\n}\n    \n    model = lgb.train(\n        params,\n        train_data,\n        valid_sets=[train_data, val_data],\n        num_boost_round=1000,\n        feval=rmsle_metric\n    )\n    # Store predictions\n    oof_predictions[val_idx] = model.predict(X_val)\n    \n    # Store feature importance\n    fold_importance = pd.DataFrame()\n    fold_importance['feature'] = feature_names\n    fold_importance['importance'] = model.feature_importance()\n    fold_importance['fold'] = fold\n    feature_importance_df = pd.concat([feature_importance_df, fold_importance])\n\n# Calculate overall metrics\nrmse = np.sqrt(mean_squared_error(y, oof_predictions))\nr2 = r2_score(y, oof_predictions)\nprint(f'Overall RMSE: {rmse:.5f}')\nprint(f'Overall R2: {r2:.5f}')\n# Calculate and print RMSLE\ny_true_log = np.log(np.clip(y, a_min=0, a_max=None) + 1)\ny_pred_log = np.log(np.clip(oof_predictions, a_min=0, a_max=None) + 1)\nrmsle = np.sqrt(mean_squared_error(y_true_log, y_pred_log))\nprint(f'Overall RMSLE: {rmsle:.5f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T02:47:23.733115Z","iopub.execute_input":"2024-12-13T02:47:23.733533Z","iopub.status.idle":"2024-12-13T02:52:49.930438Z","shell.execute_reply.started":"2024-12-13T02:47:23.733498Z","shell.execute_reply":"2024-12-13T02:52:49.929132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Feature Importance Analysis\nplt.figure(figsize=(12, 6))\nsns.barplot(\n    data=feature_importance_df.groupby('feature')['importance'].mean().sort_values(ascending=False).head(20).reset_index(),\n    x='importance',\n    y='feature'\n)\nplt.title('Top 20 Feature Importances')\nplt.xlabel('Importance Score')\nplt.tight_layout()\nplt.show()\n\n# Residual Analysis\nresiduals = y - oof_predictions\nplt.figure(figsize=(15, 5))\n\n# Residual plot\nplt.subplot(131)\nplt.scatter(oof_predictions, residuals, alpha=0.5)\nplt.axhline(y=0, color='r', linestyle='--')\nplt.xlabel('Predicted Premium')\nplt.ylabel('Residuals')\nplt.title('Residual Plot')\n\n# Residual distribution\nplt.subplot(132)\nsns.histplot(residuals, kde=True)\nplt.xlabel('Residual Value')\nplt.title('Residual Distribution')\n\n# Q-Q plot\nplt.subplot(133)\nfrom scipy import stats\nstats.probplot(residuals, dist=\"norm\", plot=plt)\nplt.title('Q-Q Plot')\n\nplt.tight_layout()\nplt.show()\n\n# Additional insights\nprint(\"\\nResidual Statistics:\")\nprint(f\"Mean of residuals: {residuals.mean():.2f}\")\nprint(f\"Std of residuals: {residuals.std():.2f}\")\nprint(f\"Skewness: {residuals.skew():.2f}\")\nprint(f\"Kurtosis: {residuals.kurtosis():.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T02:55:06.185113Z","iopub.execute_input":"2024-12-13T02:55:06.185551Z","iopub.status.idle":"2024-12-13T02:55:19.427333Z","shell.execute_reply.started":"2024-12-13T02:55:06.185514Z","shell.execute_reply":"2024-12-13T02:55:19.426149Z"}},"outputs":[],"execution_count":null}]}