{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30824,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import Libraries\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.experimental import enable_iterative_imputer  # noqa\nfrom sklearn.impute import IterativeImputer\nfrom matplotlib.ticker import FuncFormatter\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.model_selection import train_test_split, KFold, StratifiedKFold, GridSearchCV, RandomizedSearchCV\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score, accuracy_score, confusion_matrix\nimport lightgbm as lgb\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom catboost import CatBoostRegressor, Pool\nfrom scipy import stats\nfrom matplotlib.lines import Line2D\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nfrom catboost import CatBoostRegressor\n# ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:10:03.879094Z","iopub.execute_input":"2024-12-25T06:10:03.879327Z","iopub.status.idle":"2024-12-25T06:10:09.481055Z","shell.execute_reply.started":"2024-12-25T06:10:03.879306Z","shell.execute_reply":"2024-12-25T06:10:09.480376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nfrom IPython.display import display, HTML\n\n# Function to style tables\ndef style_table(df):\n    styled_df = df.style.set_table_styles([ \n        {\"selector\": \"th\", \"props\": [(\"color\", \"white\"), (\"background-color\", \"#2E3B4E\")]}  # Deep blue background for headers\n    ]).set_properties(**{\"text-align\": \"center\"}).hide(axis=\"index\")\n    return styled_df.to_html()\n\n# Function to generate random shades of color\ndef generate_random_color():\n    color = \"#{:02x}{:02x}{:02x}\".format(\n        random.randint(50, 150),\n        random.randint(50, 150),\n        random.randint(150, 255)\n    )\n    return color\n\n# Function to create styled heading with emojis and different colors for main and sub-headings\ndef styled_heading(text, background_color, text_color='white', border_color=None, font_size='30px', border_style='dashed'):\n    border_color = border_color if border_color else background_color\n    return f\"\"\"\n    <div style=\"\n        text-align: center;\n        background: {background_color};\n        color: {text_color};\n        padding: 15px;\n        font-size: {font_size};\n        font-weight: bold;\n        line-height: 1;\n        border-radius: 20px 20px 0 0;\n        margin-bottom: 20px;\n        box-shadow: 0px 4px 6px rgba(0, 0, 0, 0.2);\n        border: 3px {border_style} {border_color};\n    \">\n        {text}\n    </div>\n    \"\"\"\n\n# Define a new attractive color palette with deep blue, purple tones\nmain_heading_colors = ['#4B0082', '#283593', '#1A237E', '#512DA8', '#7B1FA2']  # Purple tones\nsub_heading_colors = ['#1976D2', '#4CAF50', '#009688', '#0288D1', '#8E24AA']  # Deep blue and purple tones\nheadings_border_color = '#FFB300'  # Darker mustard color for outer border\n\ndef print_dataset_analysis(dataset, dataset_name, n_top=5, palette_index=0):\n    heading_color = main_heading_colors[palette_index % len(main_heading_colors)]\n    sub_heading_color = sub_heading_colors[palette_index % len(sub_heading_colors)]\n    \n    # Main heading with emoji\n    heading = styled_heading(f\"📊 {dataset_name} Overview\", heading_color, 'white', border_color=headings_border_color, font_size='35px', border_style='solid')\n    display(HTML(heading))\n    \n    # Sub-headings with emojis\n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>🔍 Shape of the Dataset</h2>\"))\n    display(HTML(f\"<p>{dataset.shape[0]} rows and {dataset.shape[1]} columns</p>\"))\n    \n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>👀 First 5 Rows</h2>\"))\n    display(HTML(style_table(dataset.head(n_top))))\n    \n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>📈 Summary Statistics</h2>\"))\n    display(HTML(style_table(dataset.describe())))\n    \n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>🚨 Null Values</h2>\"))\n    null_counts = dataset.isnull().sum()\n    if null_counts.sum() == 0:\n        display(HTML(\"<p>No null values found.</p>\"))\n    else:\n        null_columns = null_counts[null_counts > 0]\n        null_columns_df = null_columns.to_frame(name='Null Values')\n        null_columns_df['Column Names with Nulls'] = null_columns.index\n        display(HTML(style_table(null_columns_df)))\n    \n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>🔍 Duplicate Rows</h2>\"))\n    duplicate_count = dataset.duplicated().sum()\n    display(HTML(f\"<p>{duplicate_count} duplicate rows found.</p>\"))\n    \n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>📝 Data Types</h2>\"))\n    dtypes_table = pd.DataFrame({\n        'Data Type': [dataset[col].dtype for col in dataset.columns],\n        'Column Name': dataset.columns\n    })\n    display(HTML(style_table(dtypes_table)))\n\n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>📋 Column Names</h2>\"))\n    display(HTML(f\"<p>{', '.join(dataset.columns)}</p>\"))\n\n    display(HTML(f\"<h2 style='font-size: 24px; color: {sub_heading_color};'>🔢 Unique Values</h2>\"))\n    unique_values_table = pd.DataFrame({\n        'Data Type': [dataset[col].dtype for col in dataset.columns],\n        'Column Name': dataset.columns,\n        'Unique Values': [len(dataset[col].unique()) for col in dataset.columns]\n    })\n    display(HTML(style_table(unique_values_table)))\n\n# Load datasets with new names\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsample_sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\n# Use different palette colors for different datasets\nprint_dataset_analysis(df_train, \"Training Data\", palette_index=0)\nprint_dataset_analysis(df_test, \"Test Data\", palette_index=1)\nprint_dataset_analysis(sample_sub, \"Sample Submission\", palette_index=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:10:27.915038Z","iopub.execute_input":"2024-12-25T06:10:27.915336Z","iopub.status.idle":"2024-12-25T06:10:41.879004Z","shell.execute_reply.started":"2024-12-25T06:10:27.915314Z","shell.execute_reply":"2024-12-25T06:10:41.878304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze value counts for 'Gender'\ngender_counts = df_train['Gender'].value_counts()\n\nprint(\"Gender Counts:\")\nprint(gender_counts)\n\n# Analyze the range of 'Age'\nage_min = df_train['Age'].min()\nage_max = df_train['Age'].max()\nprint(\"====================================\")\nprint(\"\\nAge Range:\")\nprint(f\"Minimum Age: {age_min}\")\nprint(f\"Maximum Age: {age_max}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:10:47.624645Z","iopub.execute_input":"2024-12-25T06:10:47.624973Z","iopub.status.idle":"2024-12-25T06:10:47.71737Z","shell.execute_reply.started":"2024-12-25T06:10:47.624948Z","shell.execute_reply":"2024-12-25T06:10:47.716371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a figure with two subplots: one for 'Gender' and one for 'Age'\nfig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\n# Refined color palette for Gender (Male: Vivid Amber, Female: Charcoal Blue)\npie_colors = ['#F4A300', '#003B49']  # Vivid Amber and Charcoal Blue colors\n\n# Plot for 'Gender' value counts with refined colors\ngender_counts = df_train['Gender'].value_counts()\nsns.barplot(x=gender_counts.index, y=gender_counts.values, \n            palette=pie_colors, ax=axes[0])  # Applying the custom color palette\n\n# Customize the 'Gender' plot\naxes[0].set_title('Gender Distribution', fontsize=16, fontweight='bold', color='darkblue')\naxes[0].set_xlabel('Gender', fontsize=12, color='black')\naxes[0].set_ylabel('Count', fontsize=12, color='black')\naxes[0].tick_params(axis='x', rotation=0, labelcolor='black')\naxes[0].tick_params(axis='y', labelcolor='black')\n\n# Add count annotations on the bars with white text\nfor p in axes[0].patches:\n    axes[0].annotate(f'{p.get_height():,.0f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                     ha='center', va='center', fontsize=12, color='black', fontweight='bold')\n\n# Custom colors for the Age plot\nage_color = '#32CD32'  # Lime Green color for the histogram\nage_annotation_color = 'darkred'  # White text for annotations\n\n# Plot for 'Age' range (min and max)\nsns.histplot(df_train['Age'], bins=15, kde=False, color=age_color, ax=axes[1])\n\n# Customize the 'Age' plot\naxes[1].set_title('Age Range Distribution', fontsize=16, fontweight='bold', color='darkblue')\naxes[1].set_xlabel('Age', fontsize=12, color='black')\naxes[1].set_ylabel('Count', fontsize=12, color='black')\naxes[1].tick_params(axis='x', labelcolor='black')\naxes[1].tick_params(axis='y', labelcolor='black')\n\n# Annotate Age Range with custom text color\nage_min = df_train['Age'].min()\nage_max = df_train['Age'].max()\naxes[1].annotate(f'Min: {age_min}\\nMax: {age_max}', xy=(0.5, 0.9), xycoords='axes fraction', \n                 ha='center', va='center', fontsize=14, fontweight='bold', color=age_annotation_color)\n\n# Adjust layout for better spacing\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:10:50.534955Z","iopub.execute_input":"2024-12-25T06:10:50.535284Z","iopub.status.idle":"2024-12-25T06:10:51.87718Z","shell.execute_reply.started":"2024-12-25T06:10:50.535261Z","shell.execute_reply":"2024-12-25T06:10:51.876301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Proportion of each gender in the dataset\ngender_proportion = df_train['Gender'].value_counts(normalize=True) * 100\n\nprint(\"Proportion of Each Gender in the Dataset:\")\nprint(gender_proportion)\n# Check if the dataset has a gender imbalance\nmost_common_gender = gender_proportion.idxmax()\nimbalance_percentage = gender_proportion.max() - gender_proportion.min()\n\nprint(f\"The most common gender is '{most_common_gender}' with a {gender_proportion.max():.2f}% share.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:10:57.420976Z","iopub.execute_input":"2024-12-25T06:10:57.421326Z","iopub.status.idle":"2024-12-25T06:10:57.500275Z","shell.execute_reply.started":"2024-12-25T06:10:57.421298Z","shell.execute_reply":"2024-12-25T06:10:57.499471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate gender proportions\ngender_proportion = df_train['Gender'].value_counts(normalize=True) * 100\n\n# Create a figure with two subplots: one for Pie chart and one for Bar plot\nfig, axes = plt.subplots(1, 2, figsize=(16, 8))\n\n# Define a refined color palette for both plots\npie_colors = ['#F4A300', '#003B49']  # Vivid Amber and Charcoal Blue colors\n\n# Plot 1: Pie chart for gender proportions with enhanced visuals\nwedges, texts, autotexts = axes[0].pie(gender_proportion, labels=gender_proportion.index, autopct='%1.1f%%', \n                                      colors=pie_colors, startangle=90, \n                                      wedgeprops={'edgecolor': 'white', 'linewidth': 2, 'linestyle': 'solid'}, \n                                      shadow=True, textprops={'color': 'white'})  # Set text inside pie chart to white\n\n# Add legend for the Pie chart\naxes[0].legend(wedges, gender_proportion.index, title=\"Gender\", loc=\"center left\", bbox_to_anchor=(1, 0.5), fontsize=12)\n\n# Customize Pie chart\naxes[0].set_title('Gender Proportion', fontsize=18, fontweight='bold', color='darkblue', pad=20)\naxes[0].axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\n\n# Plot 2: Bar plot for gender proportions with improved style\nsns.barplot(x=gender_proportion.index, y=gender_proportion.values, \n            palette=pie_colors, ax=axes[1])\n\n# Customize Bar plot\naxes[1].set_title('Gender Proportion Bar Plot', fontsize=18, fontweight='bold', color='darkblue', pad=20)\naxes[1].set_xlabel('Gender', fontsize=14, color='darkblue')\naxes[1].set_ylabel('Proportion (%)', fontsize=14, color='darkblue')\naxes[1].tick_params(axis='x', rotation=0, labelcolor='black', labelsize=12)\naxes[1].tick_params(axis='y', labelcolor='black', labelsize=12)\n\n# Add percentage annotations on the bars with improved styling\nfor p in axes[1].patches:\n    height = p.get_height()\n    axes[1].annotate(f'{height:.1f}%', \n                     (p.get_x() + p.get_width() / 2., height),\n                     ha='center', va='center', fontsize=14, color='black', fontweight='bold')\n\n# Add gridlines to the bar plot for better readability\naxes[1].grid(axis='y', linestyle='--', alpha=0.7)\n\n# Adjust layout for better spacing and make the plot more cohesive\nplt.tight_layout(pad=5)\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:10:59.946045Z","iopub.execute_input":"2024-12-25T06:10:59.946349Z","iopub.status.idle":"2024-12-25T06:11:00.332299Z","shell.execute_reply.started":"2024-12-25T06:10:59.946326Z","shell.execute_reply":"2024-12-25T06:11:00.331433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate mean and median age by gender\nmean_age_by_gender = df_train.groupby('Gender')['Age'].mean()\nmedian_age_by_gender = df_train.groupby('Gender')['Age'].median()\n\nprint(\"Mean Age by Gender:\")\nprint(mean_age_by_gender)\nprint(\"\\nMedian Age by Gender:\")\nprint(median_age_by_gender)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:03.924571Z","iopub.execute_input":"2024-12-25T06:11:03.92487Z","iopub.status.idle":"2024-12-25T06:11:04.077337Z","shell.execute_reply.started":"2024-12-25T06:11:03.924847Z","shell.execute_reply":"2024-12-25T06:11:04.076477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate mean and median age by gender\nmean_age_by_gender = df_train.groupby('Gender')['Age'].mean()\nmedian_age_by_gender = df_train.groupby('Gender')['Age'].median()\n\n# Create a figure with two rows: the first row for bar plots and the second row for pie charts\nfig, axes = plt.subplots(2, 2, figsize=(16, 14))\n\n# Define a refined color palette for both plots\nbar_colors = ['#F4A300', '#003B49']  # Vivid Amber and Charcoal Blue colors for bar plots\npie_colors = ['#F4A300', '#003B49']  # Vivid Amber and Charcoal Blue colors for pie charts\n\n# Plot 1: Bar plot for mean age by gender\nsns.barplot(x=mean_age_by_gender.index, y=mean_age_by_gender.values, \n            palette=bar_colors, ax=axes[0, 0])\n\n# Customize Mean Age plot\naxes[0, 0].set_title('Mean Age by Gender', fontsize=18, fontweight='bold', color='darkblue', pad=20)\naxes[0, 0].set_xlabel('Gender', fontsize=14, color='darkblue')\naxes[0, 0].set_ylabel('Mean Age', fontsize=14, color='darkblue')\naxes[0, 0].tick_params(axis='x', labelcolor='black', labelsize=12)\naxes[0, 0].tick_params(axis='y', labelcolor='black', labelsize=12)\n\n# Add value annotations on the bars\nfor p in axes[0, 0].patches:\n    height = p.get_height()\n    axes[0, 0].annotate(f'{height:.1f}', \n                        (p.get_x() + p.get_width() / 2., height),\n                        ha='center', va='center', fontsize=14, color='black', fontweight='bold')\n\n# Plot 2: Bar plot for median age by gender\nsns.barplot(x=median_age_by_gender.index, y=median_age_by_gender.values, \n            palette=bar_colors, ax=axes[0, 1])\n\n# Customize Median Age plot\naxes[0, 1].set_title('Median Age by Gender', fontsize=18, fontweight='bold', color='darkblue', pad=20)\naxes[0, 1].set_xlabel('Gender', fontsize=14, color='darkblue')\naxes[0, 1].set_ylabel('Median Age', fontsize=14, color='darkblue')\naxes[0, 1].tick_params(axis='x', labelcolor='black', labelsize=12)\naxes[0, 1].tick_params(axis='y', labelcolor='black', labelsize=12)\n\n# Add value annotations on the bars\nfor p in axes[0, 1].patches:\n    height = p.get_height()\n    axes[0, 1].annotate(f'{height:.1f}', \n                        (p.get_x() + p.get_width() / 2., height),\n                        ha='center', va='center', fontsize=14, color='black', fontweight='bold')\n\n# Plot 3: Pie chart for mean age by gender\naxes[1, 0].pie(mean_age_by_gender, labels=mean_age_by_gender.index, autopct='%1.1f%%', \n               colors=pie_colors, startangle=90, \n               wedgeprops={'edgecolor': 'white', 'linewidth': 2, 'linestyle': 'solid'}, \n               shadow=True, textprops={'color': 'white'})  # Set text inside pie chart to white\n\n# Customize Pie chart for Mean Age\naxes[1, 0].set_title('Mean Age by Gender (Pie)', fontsize=18, fontweight='bold', color='darkblue', pad=20)\naxes[1, 0].axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\naxes[1, 0].legend(mean_age_by_gender.index, title=\"Gender\", loc='upper right', fontsize=12)\n\n# Plot 4: Pie chart for median age by gender\naxes[1, 1].pie(median_age_by_gender, labels=median_age_by_gender.index, autopct='%1.1f%%', \n               colors=pie_colors, startangle=90, \n               wedgeprops={'edgecolor': 'white', 'linewidth': 2, 'linestyle': 'solid'}, \n               shadow=True, textprops={'color': 'white'})  # Set text inside pie chart to white\n\n# Customize Pie chart for Median Age\naxes[1, 1].set_title('Median Age by Gender (Pie)', fontsize=18, fontweight='bold', color='darkblue', pad=20)\naxes[1, 1].axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\naxes[1, 1].legend(median_age_by_gender.index, title=\"Gender\", loc='upper right', fontsize=12)\n\n# Adjust layout for better spacing and make the plot more cohesive\nplt.tight_layout(pad=5)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:05.986282Z","iopub.execute_input":"2024-12-25T06:11:05.986613Z","iopub.status.idle":"2024-12-25T06:11:06.812545Z","shell.execute_reply.started":"2024-12-25T06:11:05.986576Z","shell.execute_reply":"2024-12-25T06:11:06.811637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the range of age for each gender\nage_range_by_gender = df_train.groupby('Gender').agg({'Age': lambda x: x.max() - x.min()})\n\nprint(\"Age Range by Gender:\")\nprint(age_range_by_gender)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:10.054138Z","iopub.execute_input":"2024-12-25T06:11:10.054437Z","iopub.status.idle":"2024-12-25T06:11:10.144977Z","shell.execute_reply.started":"2024-12-25T06:11:10.054416Z","shell.execute_reply":"2024-12-25T06:11:10.144259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define age bins\nbins = [18, 25, 35, 45, 55, 64]\nlabels = ['18-25', '26-35', '36-45', '46-55', '56-64']\n\n# Create an Age Range column\ndf_train['Age Range'] = pd.cut(df_train['Age'], bins=bins, labels=labels, right=True)\n\n# Distribution of Gender within Age Ranges\nage_range_gender_dist = df_train.groupby('Age Range')['Gender'].value_counts(normalize=True)\n\n# Output the distribution\nprint(\"Gender distribution within each Age Range:\\n\", age_range_gender_dist)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:13.23655Z","iopub.execute_input":"2024-12-25T06:11:13.236868Z","iopub.status.idle":"2024-12-25T06:11:13.361797Z","shell.execute_reply.started":"2024-12-25T06:11:13.236843Z","shell.execute_reply":"2024-12-25T06:11:13.360997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define updated age bins and labels\nbins = [18, 22, 25, 30, 35, 40, 45, 50, 55, 60, 64]\nlabels = ['18-22', '23-25', '26-30', '31-35', '36-40', '41-45', '46-50', '51-55', '56-60', '61-64']\n\n# Create an Age Range column\ndf_train['Age Range'] = pd.cut(df_train['Age'], bins=bins, labels=labels, right=True)\n\n# Define the color palette for gender (darker shades)\ngender_palette = ['#B77A00', '#001F2D']  # Darker shades of Vivid Amber and Charcoal Blue\n\n# Create the figure with 2 subplots (2 rows, 1 column)\nfig, axes = plt.subplots(2, 1, figsize=(10, 14))\n\n# Histogram plot: Gender distribution within age ranges\nsns.histplot(data=df_train, x='Age Range', hue='Gender', stat='probability', common_norm=False, multiple=\"stack\", palette=gender_palette, ax=axes[0])\n\n# Add the proportion values inside the histogram bars for each gender\nfor p in axes[0].patches:\n    height = p.get_height()\n    x = p.get_x() + p.get_width() / 2  # x position of the bar\n    y = p.get_y() + height / 2  # y position of the bar\n    axes[0].annotate(f\"{height:.5f}\", (x, y), textcoords=\"offset points\", xytext=(0, 5), ha='center', fontsize=8, color='black')\n\naxes[0].set_title('Proportion of Gender Distribution within Age Ranges', fontsize=14, fontweight='bold', color='darkblue', pad=20)\naxes[0].set_xlabel('Age Range', fontsize=10, color='darkblue')\naxes[0].set_ylabel('Proportion', fontsize=10, color='darkblue')\naxes[0].tick_params(axis='x', rotation=45, labelsize=8)  # Decreased tick label size for x-axis\naxes[0].tick_params(axis='y', labelsize=8)  # Decreased tick label size for y-axis\naxes[0].legend(title='Gender', labels=['Female', 'Male'], loc='upper right', fontsize=12)\n\n# Line plot: Gender proportion trend across age ranges\n# Calculate the proportion of males and females for each age range\ngender_counts = df_train.groupby(['Age Range', 'Gender']).size().unstack(fill_value=0)\ngender_proportions = gender_counts.div(gender_counts.sum(axis=1), axis=0)\n\n# Plot the gender proportions as lines\ngender_proportions.plot(ax=axes[1], color=gender_palette, marker='o', linewidth=2)\naxes[1].set_title('Gender Proportions Across Age Ranges', fontsize=14, fontweight='bold', color='darkblue', pad=20)\naxes[1].set_xlabel('Age Range', fontsize=10, color='darkblue')\naxes[1].set_ylabel('Proportion', fontsize=10, color='darkblue')\naxes[1].tick_params(axis='x', rotation=45, labelsize=8)  # Decreased tick label size for x-axis\naxes[1].tick_params(axis='y', labelsize=8)  # Decreased tick label size for y-axis\naxes[1].legend(title='Gender', labels=['Female', 'Male'], loc='upper right', fontsize=12)\n\n# Add the proportion values above the lines in black\nfor i, age_range in enumerate(gender_proportions.index):\n    for j, gender in enumerate(gender_proportions.columns):\n        axes[1].annotate(f\"{gender_proportions.loc[age_range, gender]:.5f}\",\n                         (i, gender_proportions.loc[age_range, gender]),\n                         textcoords=\"offset points\",\n                         xytext=(0, 10),  # 10 points vertical offset\n                         ha='center', fontsize=7, color=gender_palette[j])\n\n# Adjust layout to avoid overlap\nplt.tight_layout(pad=5)\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:16.091921Z","iopub.execute_input":"2024-12-25T06:11:16.092269Z","iopub.status.idle":"2024-12-25T06:11:18.45606Z","shell.execute_reply.started":"2024-12-25T06:11:16.09224Z","shell.execute_reply":"2024-12-25T06:11:18.455178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mode of Age and Gender\nage_mode = df_train['Age'].mode()[0]\ngender_mode = df_train['Gender'].mode()[0]\n\n# Unique pair combinations of Age and Gender\nunique_age_gender_pairs = df_train[['Age', 'Gender']]\n\n# Output the results\nprint(\"Mode of 'Age':\", age_mode)\ndisplay(\"Unique combinations of 'Age' and 'Gender':\", unique_age_gender_pairs.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:22.176786Z","iopub.execute_input":"2024-12-25T06:11:22.17715Z","iopub.status.idle":"2024-12-25T06:11:22.288435Z","shell.execute_reply.started":"2024-12-25T06:11:22.177122Z","shell.execute_reply":"2024-12-25T06:11:22.287485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mode of Age and Gender\nage_mode = df_train['Age'].mode()[0]\ngender_mode = df_train['Gender'].mode()[0]\n\n# Unique pair combinations of Age and Gender\nunique_age_gender_pairs = df_train[['Age', 'Gender']]\n\n# Filter top 20 'Age' values by frequency\ntop_20_ages = df_train['Age'].value_counts().head(20).index\n\n# Create a barplot and pie chart subplot with increased figure size\nfig, axes = plt.subplots(1, 2, figsize=(25, 10))  # Increased figure size\n\n# Barplot: Mode of Age by Gender (only top 20 ages)\nsns.countplot(x='Age', hue='Gender', data=df_train[df_train['Age'].isin(top_20_ages)], ax=axes[0], palette=['#F4A300', '#003B49'])\naxes[0].set_title('Count of Mode Age by Gender', fontsize=20, fontweight='bold')  # Make title bold and larger\naxes[0].set_xlabel('Age', fontsize=16, fontweight='bold')  # Larger and bold x-label\naxes[0].set_ylabel('Count', fontsize=16, fontweight='bold')  # Larger and bold y-label\n\n# Increase font size and bold tick labels\naxes[0].tick_params(axis='x', labelsize=18, labelrotation=45)  # Larger x-tick labels\naxes[0].tick_params(axis='y', labelsize=18)  # Larger y-tick labels\n\n# Move the legend outside the plot\naxes[0].legend(title='Gender', loc='upper left', bbox_to_anchor=(1, 1), labels=['Female', 'Male'], fontsize=20)\n\n# Pie chart: Mode values for Gender (based on the mode of 'Gender' in df_train)\ngender_counts = df_train['Gender'].value_counts()\n\n# Increase font size for the pie chart labels and percentage values, and set color to white\naxes[1].pie(gender_counts, labels=gender_counts.index, autopct='%1.1f%%', startangle=90, colors=['#F4A300', '#003B49'], \n            textprops={'fontsize': 18, 'fontweight': 'bold', 'color': 'white'})  # Set text color to white\n\naxes[1].set_title('Gender Distribution (Mode Values)', fontsize=25, fontweight='bold')  # Make title bold and larger\n\n# Display the results\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:25.35878Z","iopub.execute_input":"2024-12-25T06:11:25.359097Z","iopub.status.idle":"2024-12-25T06:11:26.572416Z","shell.execute_reply.started":"2024-12-25T06:11:25.35907Z","shell.execute_reply":"2024-12-25T06:11:26.57161Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n1. According to this dataset:\n   - There are about **602,571 males** in the training dataset.\n   - There are about **597,429 females** in the training dataset.\n\n2. **Age Statistics**:\n   - Minimum age: **18.0**\n   - Maximum age: **64.0**\n   - Mean age: **41.1**\n   - Median age: **41.0**\n   - Age range: **46.0**\n\n3. **Proportion of Age Ranges by Gender**:\n   - **18-25 years**:\n     - Male: **50.14%**\n     - Female: **49.86%**\n   - **26-35 years**:\n     - Male: **50.23%**\n     - Female: **49.77%**\n   - **36-45 years**:\n     - Male: **50.26%**\n     - Female: **49.74%**\n   - **46-55 years**:\n     - Male: **50.21%**\n     - Female: **49.79%**\n   - **56-64 years**:\n     - Male: **50.20%**\n     - Female: **49.80%**\n\n4. **Highest Proportion**:\n   - The highest proportion is in the **36-45 age range**, with:\n     - Male: **50.26%**\n     - Female: **49.74%**\n\n5. **Highest Frequency by Age**:\n   - For males, the highest frequency is at age **53** with a count of **13,315**.\n   - For females, the highest frequency is at age **63** with a count of **13,123**.\n\n6. **Overall Gender Proportion**:\n   - **Females** constitute **50.2%** of the dataset.\n   - **Males** constitute **49.8%** of the dataset.\n   - This indicates that **females are present in a slightly higher proportion** than males in this dataset.\n","metadata":{}},{"cell_type":"code","source":"# Get unique values in categorical columns\nunique_gender = df_train['Gender'].unique()\nunique_marital_status = df_train['Marital Status'].unique()\n\n# Display the unique values\nprint(\"Unique values in Gender:\", unique_gender)\nprint(\"Unique values in Marital Status:\", unique_marital_status)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:33.063458Z","iopub.execute_input":"2024-12-25T06:11:33.063749Z","iopub.status.idle":"2024-12-25T06:11:33.157289Z","shell.execute_reply.started":"2024-12-25T06:11:33.063725Z","shell.execute_reply":"2024-12-25T06:11:33.156639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the frequency counts of each category in Gender and Marital Status\ngender_counts = df_train['Gender'].value_counts().reset_index()\ngender_counts.columns = ['Gender', 'Gender Count']\n\nmarital_status_counts = df_train['Marital Status'].value_counts().reset_index()\nmarital_status_counts.columns = ['Marital Status', 'Marital Status Count']\n\n# Merge the two dataframes into one\ncombined_analysis = pd.merge(gender_counts, marital_status_counts, how='cross')\n\n# Display the combined analysis\ndisplay(combined_analysis)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:35.079398Z","iopub.execute_input":"2024-12-25T06:11:35.079689Z","iopub.status.idle":"2024-12-25T06:11:35.241025Z","shell.execute_reply.started":"2024-12-25T06:11:35.079664Z","shell.execute_reply":"2024-12-25T06:11:35.240193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get summary statistics for Age and Annual Income\nage_summary = df_train['Age'].describe()\nincome_summary = df_train['Annual Income'].describe()\n\n# Display the summaries\nprint(\"Age Summary:\\n\", age_summary)\nprint(\"\\nAnnual Income Summary:\\n\", income_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:37.104515Z","iopub.execute_input":"2024-12-25T06:11:37.104821Z","iopub.status.idle":"2024-12-25T06:11:37.223484Z","shell.execute_reply.started":"2024-12-25T06:11:37.104798Z","shell.execute_reply":"2024-12-25T06:11:37.222685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the statistics for Age and Annual Income\nage_mean = df_train['Age'].mean()\nage_max = df_train['Age'].max()\nage_min = df_train['Age'].min()\n\nincome_mean = df_train['Annual Income'].mean()\nincome_max = df_train['Annual Income'].max()\nincome_min = df_train['Annual Income'].min()\n\n\n# Calculate the statistics for Age and Annual Income grouped by Gender\ngender_stats = df_train.groupby('Gender').agg({\n    'Age': ['mean', 'max', 'min'],\n    'Annual Income': ['mean', 'max', 'min']\n})\n\n# Reset the index and flatten the multi-level columns\ngender_stats_reset = gender_stats.reset_index()\ngender_stats_reset.columns = ['Gender', 'Age_mean', 'Age_max', 'Age_min', 'Annual_Income_mean', 'Annual_Income_max', 'Annual_Income_min']\n\n# Display the summary\ndisplay(gender_stats_reset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:39.243549Z","iopub.execute_input":"2024-12-25T06:11:39.243839Z","iopub.status.idle":"2024-12-25T06:11:39.37815Z","shell.execute_reply.started":"2024-12-25T06:11:39.24381Z","shell.execute_reply":"2024-12-25T06:11:39.377277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the color palette for gender (same colors for both bar plots and box plots)\ngender_palette = ['#B77A00', '#001F2D']  # Darker shades of Vivid Amber and Charcoal Blue\n\n# Create a figure with subplots (2 rows, 2 columns)\nfig, axes = plt.subplots(2, 2, figsize=(16, 12))\n\n# Bar plot for Age statistics by Gender (Mean)\nsns.barplot(\n    x='Gender', \n    y='Age_mean', \n    data=gender_stats_reset, \n    ax=axes[0, 0], \n    palette=gender_palette  # Use the same palette for the bar plot\n)\naxes[0, 0].set_title('Age Statistics by Gender (Mean)', fontsize=18, fontweight='bold')\naxes[0, 0].set_ylabel('Age', fontsize=14, fontweight='bold')\naxes[0, 0].set_xlabel('Gender', fontsize=14, fontweight='bold')\n\n# Bar plot for Annual Income statistics by Gender (Mean)\nsns.barplot(\n    x='Gender', \n    y='Annual_Income_mean', \n    data=gender_stats_reset, \n    ax=axes[0, 1], \n    palette=gender_palette  # Use the same palette for the bar plot\n)\naxes[0, 1].set_title('Annual Income Statistics by Gender (Mean)', fontsize=18, fontweight='bold')\naxes[0, 1].set_ylabel('Annual Income', fontsize=14, fontweight='bold')\naxes[0, 1].set_xlabel('Gender', fontsize=14, fontweight='bold')\n\n# Box plot for Age by Gender (Distribution)\nsns.boxplot(\n    x='Gender', \n    y='Age', \n    data=df_train, \n    ax=axes[1, 0], \n    palette=gender_palette  # Use the same palette for the box plot\n)\naxes[1, 0].set_title('Age Distribution by Gender', fontsize=18, fontweight='bold')\naxes[1, 0].set_ylabel('Age', fontsize=14, fontweight='bold')\naxes[1, 0].set_xlabel('Gender', fontsize=14, fontweight='bold')\n\n# Box plot for Annual Income by Gender (Distribution)\nsns.boxplot(\n    x='Gender', \n    y='Annual Income', \n    data=df_train, \n    ax=axes[1, 1], \n    palette=gender_palette  # Use the same palette for the box plot\n)\naxes[1, 1].set_title('Annual Income Distribution by Gender', fontsize=18, fontweight='bold')\naxes[1, 1].set_ylabel('Annual Income', fontsize=14, fontweight='bold')\naxes[1, 1].set_xlabel('Gender', fontsize=14, fontweight='bold')\n\n# Add spacing between subplots\nplt.tight_layout()\n\n# Display the visualization\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:41.309817Z","iopub.execute_input":"2024-12-25T06:11:41.310179Z","iopub.status.idle":"2024-12-25T06:11:42.91831Z","shell.execute_reply.started":"2024-12-25T06:11:41.310146Z","shell.execute_reply":"2024-12-25T06:11:42.917543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Binning and combining age and income for df_train only\nage_bins = [18, 22, 25, 30, 35, 40, 45, 50, 55, 60, 64]\nage_labels = ['18-22', '23-25', '26-30', '31-35', '36-40', '41-45', '46-50', '51-55', '56-60', '61-64']\nincome_bins = [0, 30000, 50000, 70000, 100000, 150000]\nincome_labels = ['Low', 'Medium', 'High', 'Very High', 'Top']\n\n# Assuming df_train is your training dataset\ndf_train['Age Range'] = pd.cut(df_train['Age'], bins=age_bins, labels=age_labels)\ndf_train['Income Group'] = pd.cut(df_train['Annual Income'], bins=income_bins, labels=income_labels)\n\n# Count the occurrences of each combination of Age Range and Income Group\ndf_train['Count'] = df_train.groupby(['Age Range', 'Income Group'])['Age Range'].transform('count')\n\n# Find the maximum values for Age and Income within each Age Range and Income Group\ndf_train['Max Age'] = df_train.groupby(['Age Range', 'Income Group'])['Age'].transform('max')\ndf_train['Max Income'] = df_train.groupby(['Age Range', 'Income Group'])['Annual Income'].transform('max')\n\n# Display only the new columns (Age Range, Income Group, Count, Max Age, Max Income)\ndf_train[['Age Range', 'Income Group', 'Count', 'Max Age', 'Max Income']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:45.797576Z","iopub.execute_input":"2024-12-25T06:11:45.797891Z","iopub.status.idle":"2024-12-25T06:11:45.981216Z","shell.execute_reply.started":"2024-12-25T06:11:45.797868Z","shell.execute_reply":"2024-12-25T06:11:45.98009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define custom color palettes for each plot\nage_range_palette = ['#7f278f', '#6B5B95', '#8f2727', '#F7B7A3', '#36b0d1', '#32a852', '#a83281', '#7ba832', '#ebcf52', '#27598f']  # Soft and warm colors\nincome_group_palette = ['#A6D608', '#A4B1B4', '#F4D03F', '#F39C12', '#2f278f']  # Earthy tones with pops of yellow\ngender_palette = ['#D54F39', '#34A853']  # Bright red and green for gender\nage_income_palette = ['#888f27', '#278f8f', '#8f273a', '#278f3d', '#8f2a27']  # Bright pastels for the count plot\n\n# Set up the figure and axes for multiple subplots\nfig, axes = plt.subplots(2, 2, figsize=(16, 12))\n\n# Plot 1: Barplot for Age Range vs Count with custom colors\nsns.barplot(x='Age Range', y='Count', data=df_train, ax=axes[0, 0], palette=age_range_palette)\naxes[0, 0].set_title('Age Range vs Count', fontsize=16, fontweight='bold')\naxes[0, 0].set_xlabel('Age Range', fontsize=12)\naxes[0, 0].set_ylabel('Count', fontsize=12)\n\n# Add values above bars in first plot\nfor p in axes[0, 0].patches:\n    axes[0, 0].annotate(f'{p.get_height():.0f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                        ha='center', va='center', fontsize=7, color='black', xytext=(0, 10),\n                        textcoords='offset points')\n\n# Plot 2: Barplot for Income Group vs Count with custom colors\nsns.barplot(x='Income Group', y='Count', data=df_train, ax=axes[0, 1], palette=income_group_palette)\naxes[0, 1].set_title('Income Group vs Count', fontsize=16, fontweight='bold')\naxes[0, 1].set_xlabel('Income Group', fontsize=12)\naxes[0, 1].set_ylabel('Count', fontsize=12)\n\n# Add values above bars in second plot\nfor p in axes[0, 1].patches:\n    axes[0, 1].annotate(f'{p.get_height():.0f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                        ha='center', va='center', fontsize=7, color='black', xytext=(0, 10),\n                        textcoords='offset points')\n\n# Plot 3: Scatterplot for Max Age vs Max Income based on Gender with custom colors\nsns.scatterplot(x='Max Age', y='Max Income', hue='Gender', data=df_train, ax=axes[1, 0], palette=gender_palette, style='Gender', markers=[\"o\", \"X\"], edgecolor='darkred')\naxes[1, 0].set_title('Max Age vs Max Income (Gender)', fontsize=16, fontweight='bold')\naxes[1, 0].set_xlabel('Max Age', fontsize=12)\naxes[1, 0].set_ylabel('Max Income', fontsize=12)\n\n# Place legend outside the scatter plot\naxes[1, 0].legend(title='Gender', bbox_to_anchor=(1.05, 1), loc='upper left')\n\n# Plot 4: Countplot for combinations of Age Range and Income Group with custom colors\nsns.countplot(x='Age Range', hue='Income Group', data=df_train, ax=axes[1, 1], palette=age_income_palette)\naxes[1, 1].set_title('Age Range and Income Group visualization', fontsize=16, fontweight='bold')\naxes[1, 1].set_xlabel('Age Range', fontsize=12)\naxes[1, 1].set_ylabel('Count', fontsize=12)\n\n# Place legend outside the countplot\naxes[1, 1].legend(title='Income Group', bbox_to_anchor=(1.05, 1), loc='upper left')\n\n# Adjust layout for better spacing\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:11:47.870675Z","iopub.execute_input":"2024-12-25T06:11:47.871004Z","iopub.status.idle":"2024-12-25T06:12:32.878472Z","shell.execute_reply.started":"2024-12-25T06:11:47.870974Z","shell.execute_reply":"2024-12-25T06:12:32.877568Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n### 1. Unique Values in Columns\n- The unique values present in the **Gender** column are:\n  - 'Female'\n  - 'Male'\n- The unique values present in the **Marital Status** column are:\n  - 'Married'\n  - 'Divorced'\n  - 'Single'\n\n### 2. Frequency Counts by Gender and Marital Status\n- **Males**:\n  - Single: **395,391**\n  - Married: **394,316**\n  - Divorced: **391,764**\n- **Females**:\n  - Single: **395,391**\n  - Married: **394,316**\n  - Divorced: **391,764**\n\n### 3. Age Statistics\n- Maximum age: **64**\n- Minimum age: **1.8**\n- Mean age: **41.15**\n\n### 4. Annual Income Statistics\n- Maximum annual income: **149,997**\n- Minimum annual income: **1**\n- Mean annual income: **32,745.22**\n- **Mean Annual Income by Gender**:\n  - Female: **32,776.0**\n  - Male: **32,714.7**\n\n### 5. Gender Age Statistics\n- Maximum age of females: **54-55**\n- Maximum age of males: **53-54**\n- Maximum frequency count by age: **56-60** (count: **50,908**)\n- Minimum frequency count by age: **23-25** (count: **29,149**)\n\n### 6. Income Group Distribution\n- **Low Income Group**:\n  - Highest count: **68,872**\n  - Lowest count: **6,563**\n- **Medium Income Group**: **21,442**\n- **High Income Group**: **8,590**\n- **Very High Income Group**: **8,467**\n\n### 7. Income by Age Group\n- Maximum income:\n  - Age range: **56-60**\n  - Income group: **Low Income**\n- Minimum income:\n  - **Male**: Ages **22, 30-35, 50**\n  - **Female**: Ages **25, 40, 45, 55, 60, 65**\n- **Highest Income by Gender**:\n  - Males at age **22** have the maximum income: **149,993.0**\n  - Females in age ranges **30-40** and **50-64** have the maximum income.\n\n### 8. Top Income Group by Age Range\n- **56-60**: Maximum income group.\n- **31-35, 36-40, 46-50, 51-55**: Highest earners belong to the **Medium Income Group**.\n","metadata":{}},{"cell_type":"code","source":"# Count the occurrences of each Number of Dependents for each Education Level\ndependents_distribution = df_train.groupby('Education Level')['Number of Dependents'].value_counts(normalize=True).unstack()\n\n# Rename columns for clarity\ndependents_distribution.columns.name = \"Number of Dependents\"\ndependents_distribution.fillna(0, inplace=True)\n\n# Output the distribution\ndisplay(\"Distribution of Number of Dependents by Education Level:\\n\", dependents_distribution)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:12:54.607658Z","iopub.execute_input":"2024-12-25T06:12:54.608002Z","iopub.status.idle":"2024-12-25T06:12:54.72996Z","shell.execute_reply.started":"2024-12-25T06:12:54.607973Z","shell.execute_reply":"2024-12-25T06:12:54.729094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up a figure with 2 rows and 2 columns of subplots, with increased figsize\nfig, axes = plt.subplots(2, 2, figsize=(25, 16))  # Increased figure size for better prominence\n\n# Plot 1: Grouped bar plot for Number of Dependents by Education Level\ndependents_distribution.plot(kind='bar', ax=axes[0, 0], color=sns.color_palette('Set2', len(dependents_distribution.columns)))\naxes[0, 0].set_title('Number of Dependents by Education Level', fontsize=20, fontweight='bold', color='darkblue')\naxes[0, 0].set_xlabel('Education Level', fontsize=16, fontweight='bold', color='darkblue')\naxes[0, 0].set_ylabel('Proportion', fontsize=16, fontweight='bold', color='darkblue')\naxes[0, 0].legend(title='Number of Dependents', fontsize=14, title_fontsize=16, loc='upper left', bbox_to_anchor=(1.05, 1),\n                  frameon=True, shadow=True)\naxes[0, 0].tick_params(axis='x', labelsize=14, labelrotation=45, labelcolor='black')\naxes[0, 0].tick_params(axis='y', labelsize=14, labelcolor='black')\n\n# Plot 2: Pie chart for proportions of Education Levels\neducation_level_proportions = dependents_distribution.sum(axis=1) / dependents_distribution.sum().sum()  # Calculate the proportions\naxes[0, 1].pie(education_level_proportions, labels=education_level_proportions.index, \n               autopct='%1.1f%%', startangle=90, colors=sns.color_palette('tab20', len(education_level_proportions)),\n               wedgeprops={'edgecolor': 'black', 'linewidth': 1, 'linestyle': 'solid'}, radius=0.8, labeldistance=1.05)\naxes[0, 1].set_title('Proportion of Education Levels', fontsize=20, fontweight='bold', color='darkred')\naxes[0, 1].legend(title='Education Level', fontsize=14, title_fontsize=16, loc='upper left', bbox_to_anchor=(1.05, 1),\n                  frameon=True, shadow=True)\n\n# Plot 3: Violin plot to visualize the distribution of proportions\nsns.violinplot(data=dependents_distribution, ax=axes[1, 0], palette='coolwarm', linewidth=1.5)\naxes[1, 0].set_title('Distribution of Dependents Proportions', fontsize=20, fontweight='bold', color='green')\naxes[1, 0].set_xlabel('Number of Dependents', fontsize=16, fontweight='bold', color='green')\naxes[1, 0].set_ylabel('Proportion', fontsize=16, fontweight='bold', color='green')\naxes[1, 0].set_xticks(range(len(dependents_distribution.columns)))\naxes[1, 0].set_xticklabels(dependents_distribution.columns, fontsize=14, rotation=45, fontweight='bold', color='black')\naxes[1, 0].tick_params(axis='y', labelsize=14, labelcolor='black')\naxes[1, 0].grid(True, linestyle='--', alpha=0.7)\n\n# Plot 4: Scatter plot for proportions with Education Levels as categories\nfor level in dependents_distribution.index:\n    sns.scatterplot(x=dependents_distribution.columns, y=dependents_distribution.loc[level],\n                    label=level, ax=axes[1, 1], s=150, alpha=0.8, marker='o', edgecolor='black', linewidth=1.5)\naxes[1, 1].set_title('Scatter Plot of Dependents Distribution', fontsize=20, fontweight='bold', color='purple')\naxes[1, 1].set_xlabel('Number of Dependents', fontsize=16, fontweight='bold', color='purple')\naxes[1, 1].set_ylabel('Proportion', fontsize=16, fontweight='bold', color='purple')\naxes[1, 1].legend(title='Education Level', fontsize=14, title_fontsize=16, loc='upper left', bbox_to_anchor=(1.05, 1),\n                  frameon=True, shadow=True)\naxes[1, 1].tick_params(axis='x', labelsize=14, labelrotation=45, labelcolor='black')\naxes[1, 1].tick_params(axis='y', labelsize=14, labelcolor='black')\n\n# Add borders to the plots for emphasis\nfor ax in axes.flat:\n    ax.spines['top'].set_visible(False)\n    ax.spines['right'].set_visible(False)\n    ax.spines['left'].set_color('black')\n    ax.spines['bottom'].set_color('black')\n    ax.spines['left'].set_linewidth(1.5)\n    ax.spines['bottom'].set_linewidth(1.5)\n\n# Adjust layout for better spacing\nplt.tight_layout(pad=3.0)  # Increase padding to avoid overlap\n\n# Show the plots\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:12:57.303847Z","iopub.execute_input":"2024-12-25T06:12:57.304197Z","iopub.status.idle":"2024-12-25T06:12:58.519249Z","shell.execute_reply.started":"2024-12-25T06:12:57.304168Z","shell.execute_reply":"2024-12-25T06:12:58.518317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the average and most common number of dependents for each education level\ndependents_summary = df_train.groupby('Education Level')['Number of Dependents'].agg(\n    Average='mean',\n    Most_Common=lambda x: x.value_counts().idxmax()\n).reset_index()\n\n# Display the summarized DataFrame\ndisplay(\"Summary of Dependents by Education Level:\", dependents_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:01.905686Z","iopub.execute_input":"2024-12-25T06:13:01.905994Z","iopub.status.idle":"2024-12-25T06:13:02.020938Z","shell.execute_reply.started":"2024-12-25T06:13:01.905969Z","shell.execute_reply":"2024-12-25T06:13:02.020226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the color palette for gender (same colors for both bar plots and pie charts)\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']  # Darker shades of Vivid Amber and Charcoal Blue\n\n# Create a figure with 1 row and 2 columns for the pie chart and bar plot\nfig, axes = plt.subplots(1, 2, figsize=(16, 7))\n\n# Plot 1: Pie chart showing the average number of dependents by Education Level\ndependents_summary.set_index('Education Level')['Average'].plot(kind='pie', autopct='%1.1f%%', ax=axes[0], \n                                                              colors=gender_palette[:len(dependents_summary)],\n                                                              legend=False, textprops={'color': 'white'})  # Set text color to white\naxes[0].set_title('Average Number of Dependents by Education Level', fontsize=16, fontweight='bold')\naxes[0].set_ylabel('')  # Remove y-axis label for pie chart\n\n# Plot 2: Bar plot showing the most common number of dependents by Education Level\nbar_plot = sns.barplot(x='Education Level', y='Most_Common', data=dependents_summary, ax=axes[1], \n                       palette=gender_palette[:len(dependents_summary)])\n\n# Add values above bars with black text\nfor p in bar_plot.patches:\n    bar_plot.annotate(f'{p.get_height():.0f}', \n                      (p.get_x() + p.get_width() / 2., p.get_height()), \n                      ha='center', va='center', fontsize=12, color='black', \n                      xytext=(0, 10), textcoords='offset points')\n\naxes[1].set_title('Most Common Number of Dependents by Education Level', fontsize=16, fontweight='bold')\naxes[1].set_xlabel('Education Level', fontsize=12)\naxes[1].set_ylabel('Most Common Number of Dependents', fontsize=12)\n\n# Create a custom legend to match the color palette\ncustom_legend_handles = [plt.Line2D([0], [0], marker='o', color='w', markerfacecolor=color, markersize=10, label=name)\n                         for color, name in zip(gender_palette[:len(dependents_summary)], dependents_summary['Education Level'])]\nfig.legend(handles=custom_legend_handles, title=\"Education Level\", fontsize=12, title_fontsize=14, loc='lower center', ncol=4)\n\n# Adjust layout for better spacing\nplt.tight_layout(rect=[0, 0.1, 1, 1])  # Leave space for legend at the bottom\n\n# Show the plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:04.423699Z","iopub.execute_input":"2024-12-25T06:13:04.42399Z","iopub.status.idle":"2024-12-25T06:13:04.872759Z","shell.execute_reply.started":"2024-12-25T06:13:04.423969Z","shell.execute_reply":"2024-12-25T06:13:04.87195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate mean and median income for each combination of Number of Dependents and Education Level\nincome_analysis = df_train.groupby(['Education Level', 'Number of Dependents'])['Annual Income'].agg(['mean', 'max']).reset_index()\n\n# Output income analysis\nprint(\"Income Analysis by Education Level and Number of Dependents:\\n\", income_analysis)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:07.625538Z","iopub.execute_input":"2024-12-25T06:13:07.625902Z","iopub.status.idle":"2024-12-25T06:13:07.756834Z","shell.execute_reply.started":"2024-12-25T06:13:07.625872Z","shell.execute_reply":"2024-12-25T06:13:07.756096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate mean and max income for each combination of Number of Dependents and Education Level\nincome_analysis = df_train.groupby(['Education Level', 'Number of Dependents'])['Annual Income'].agg(['mean', 'max']).reset_index()\n\n# Define the color palette for gender (same colors for both bar plots and pie charts)\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']  # Darker shades of Vivid Amber and Charcoal Blue\n\n# Create a figure with 2 rows and 1 column for the bar plot and pie chart (row-wise arrangement)\nfig, axes = plt.subplots(2, 1, figsize=(20, 18))  # Increased figsize for better prominence\n\n# Plot 1: Bar plot showing mean and max income for each combination of Education Level and Number of Dependents\nsns.barplot(x='Number of Dependents', y='mean', hue='Education Level', data=income_analysis, ax=axes[0], palette=gender_palette)\naxes[0].set_title('Mean Annual Income by Education Level and Number of Dependents', fontsize=22, fontweight='bold')\naxes[0].set_xlabel('Number of Dependents', fontsize=18)\naxes[0].set_ylabel('Mean Annual Income', fontsize=18)\n\n# Add values above the bars with black color and increase font size\nfor p in axes[0].patches:\n    axes[0].annotate(f'{p.get_height():,.0f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                     ha='center', va='center', fontsize=9, color='black', fontweight='bold', xytext=(0, 5),\n                     textcoords='offset points')\n\n# Move the legend outside of the bar plot with larger font size\naxes[0].legend(title='Education Level', fontsize=14, title_fontsize=16, bbox_to_anchor=(1.05, 1), loc='upper left')\n\n# Plot 2: Pie chart showing the distribution of mean income by Education Level\neducation_income = income_analysis.groupby('Education Level')['mean'].mean()\neducation_income.plot(kind='pie', autopct='%1.1f%%', ax=axes[1], colors=gender_palette,\n                      legend=False, textprops={'color': 'white', 'fontsize': 14}, radius=1.2)  # Increased radius for larger pie chart\naxes[1].set_title('Mean Annual Income Distribution by Education Level', fontsize=22, fontweight='bold')\naxes[1].set_ylabel('')  # Remove y-axis label for pie chart\n\n# Adjust layout for better spacing\nplt.tight_layout(pad=5.0)  # Increased padding for better spacing between plots\n\n# Show the plots\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:08.799737Z","iopub.execute_input":"2024-12-25T06:13:08.800091Z","iopub.status.idle":"2024-12-25T06:13:09.573622Z","shell.execute_reply.started":"2024-12-25T06:13:08.80006Z","shell.execute_reply":"2024-12-25T06:13:09.572833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a threshold for high-dependency (e.g., 3 or more dependents)\nhigh_dependency_threshold = 3\n\n# Calculate the proportion of individuals with high dependency for each Education Level\nhigh_dependency_profile = df_train[df_train['Number of Dependents'] >= high_dependency_threshold].groupby('Education Level').size() / df_train.groupby('Education Level').size()\n\n# Output high-dependency profile\nprint(\"Proportion of High-Dependency Individuals by Education Level:\\n\", high_dependency_profile)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:13.56391Z","iopub.execute_input":"2024-12-25T06:13:13.564263Z","iopub.status.idle":"2024-12-25T06:13:13.7579Z","shell.execute_reply.started":"2024-12-25T06:13:13.564235Z","shell.execute_reply":"2024-12-25T06:13:13.756964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a threshold for high dependency\nhigh_dependency_threshold = 3\n\n# Calculate the proportion of individuals with high dependency for each Education Level\nhigh_dependency_profile = df_train[df_train['Number of Dependents'] >= high_dependency_threshold].groupby('Education Level').size() / df_train.groupby('Education Level').size()\n\n# Define the color palette for gender (same colors for both bar plots and pie charts)\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']  # Darker shades of Vivid Amber and Charcoal Blue\n\n# Create a figure with 2 subplots (bar plot and pie chart)\nfig, axes = plt.subplots(1, 2, figsize=(30, 15))  # Increased width for side-by-side arrangement\n\n# Plot 1: Bar plot showing the proportion of high-dependency individuals by Education Level\nsns.barplot(x=high_dependency_profile.index, y=high_dependency_profile.values, ax=axes[0], palette=gender_palette)\naxes[0].set_title('Proportion of High-Dependency Individuals by Education Level', fontsize=24, fontweight='bold')\naxes[0].set_xlabel('Education Level', fontsize=20)\naxes[0].set_ylabel('Proportion of High-Dependency', fontsize=20)\n\n# Make ticks more prominent\naxes[0].tick_params(axis='x', labelsize=18, labelrotation=45, width=3, colors='black')  # x-axis ticks\naxes[0].tick_params(axis='y', labelsize=18, width=3, colors='black')  # y-axis ticks\n\n# Add values above the bars with black color and increase font size\nfor p in axes[0].patches:\n    axes[0].annotate(f'{p.get_height():.2f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                     ha='center', va='center', fontsize=20, color='black', fontweight='bold', xytext=(0, 5),\n                     textcoords='offset points')\n\n# Plot 2: Pie chart showing the proportion of high-dependency individuals by Education Level\nhigh_dependency_profile.plot(kind='pie', autopct='%1.1f%%', ax=axes[1], colors=gender_palette,\n                             legend=False, textprops={'color': 'white', 'fontsize': 18}, radius=1.2)  # Increased radius for larger pie chart\naxes[1].set_title('Proportion of High-Dependency Individuals by Education Level', fontsize=24, fontweight='bold')\naxes[1].set_ylabel('')  # Remove y-axis label for pie chart\n\n# Create a custom legend to match the color palette\ncustom_legend_handles = [\n    plt.Line2D([0], [0], marker='o', color='w', markerfacecolor=color, markersize=15, label=name)\n    for color, name in zip(gender_palette, high_dependency_profile.index)\n]\nfig.legend(handles=custom_legend_handles, title=\"Education Level\", fontsize=18, title_fontsize=20, loc='lower center', ncol=4)\n\n# Adjust layout for better spacing\nplt.tight_layout(rect=[0, 0.1, 1, 1])  # Leave space for legend at the bottom\n\n# Show the plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:15.495839Z","iopub.execute_input":"2024-12-25T06:13:15.496198Z","iopub.status.idle":"2024-12-25T06:13:16.366461Z","shell.execute_reply.started":"2024-12-25T06:13:15.496166Z","shell.execute_reply":"2024-12-25T06:13:16.365655Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n### 1. Educational Level and Frequency Count\n- People with **High School** education level have:\n  - The highest frequency count: **0.203571** (3 Number of Dependents).\n  - The lowest frequency count: **0.196724** (1 Number of Dependents).\n- People with **Master's** education level have a frequency count of **0.198223**.\n\n### 2. Educational Level and Proportion by Number of Dependents\n- **Bachelor's** education level:\n  - Highest proportion: **>0.2** (3 Number of Dependents).\n- **Bachelor's and PhD** education levels:\n  - Highest proportion: **0.2–0.4** (3 & 4 Number of Dependents).\n- **High School and Master's** education levels:\n  - Highest proportion: **0.2–0.4** (3 Number of Dependents).\n- All educational levels have an equal overall proportion of about **25%** for Bachelor's, Master's, High School, and PhD.\n\n### 3. Proportion by Number of Dependents\n- **Highest Proportion**:\n  - Number of Dependents: **3**\n  - Proportion: **0.3–0.5**\n- **Lowest Proportion**:\n  - Number of Dependents: **1**\n  - Proportion: **0.1451**\n- **Other Proportions**:\n  - Dependents 0: **0.197–0.208**\n  - Dependents 2: **0.195–0.199**\n\n### 4. Average and Most Common Number of Dependents by Educational Level\n- **Bachelor's**:\n  - Average: **2.011277**\n  - Most common: **4.0**\n- **High School**:\n  - Average: **2.007098**\n  - Most common: **3.0**\n- **Master's**:\n  - Average: **2.009063**\n  - Most common: **3.0**\n- **PhD**:\n  - Average: **2.012166**\n  - Most common: **3.0**\n\n### 5. Income Statistics by Educational Level and Number of Dependents\n- **Highest Mean Income**:\n  - Educational Level: **High School**\n  - Number of Dependents: **4**\n  - Average Income: **33,014.09**\n- **Lowest Mean Income**:\n  - Educational Level: **PhD**\n  - Number of Dependents: **0**\n  - Average Income: **32,439.47**\n- **Maximum Income**:\n  - Educational Level: **High School**\n  - Number of Dependents: **4**\n  - Income: **149,997.0**\n- **Minimum Income**:\n  - Educational Levels: **Bachelor's & High School**\n  - Number of Dependents: **4**\n  - Income: **149,992.0**\n- **Maximum Mean Income**:\n  - Educational Level: **High School**\n  - Number of Dependents: **3**\n  - Mean Income: **32,930**\n- **Minimum Mean Income**:\n  - Educational Level: **Master's**\n  - Number of Dependents: **2**\n  - Mean Income: **32,519**\n\n### 6. Proportion of People by Educational Level\n- **Bachelor's**: **0.368247**\n- **High School**: **0.368313**\n- **Master's**: **0.367447**\n- **PhD**: **0.368716**\n\n### 7. Key Insight\n- The highest proportion of people is among those with a **Master's education level**.\n","metadata":{}},{"cell_type":"code","source":"# Calculate the mean, max, min, and count of Health Score for each Occupation\nhealth_score_by_occupation = df_train.groupby('Occupation')['Health Score'].agg(['mean', 'max', 'min', 'count'])\n\n# Output the selected statistics (mean, max, min, count) of Health Score by Occupation\nprint(\"Health Score by Occupation:\\n\", health_score_by_occupation)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:21.299135Z","iopub.execute_input":"2024-12-25T06:13:21.29944Z","iopub.status.idle":"2024-12-25T06:13:21.412211Z","shell.execute_reply.started":"2024-12-25T06:13:21.299414Z","shell.execute_reply":"2024-12-25T06:13:21.411469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute missing values in 'Occupation' with the mode (most frequent value)\noccupation_imputer = SimpleImputer(strategy='most_frequent')\n\n# Ensure the input to fit_transform is a 2D array by selecting the column as DataFrame\ndf_train['Occupation'] = occupation_imputer.fit_transform(df_train[['Occupation']]).flatten()\ndf_test['Occupation'] = occupation_imputer.transform(df_test[['Occupation']]).flatten()\n\n# Define the custom color palette\ngender_palette = ['#B77A00', '#001F2D', '#29002d']\n\n# Ensure the palette has enough colors for the number of unique occupations by cycling the colors\noccupation_palette = gender_palette * (len(df_train['Occupation'].unique()) // len(gender_palette)) + gender_palette[:len(df_train['Occupation'].unique()) % len(gender_palette)]\n\n# Create a dictionary that maps occupations to the colors in the palette\noccupation_color_map = dict(zip(df_train['Occupation'].unique(), occupation_palette))\n\n# Assign navy blue to \"Unemployed\"\noccupation_color_map['Unemployed'] = '#001F2D'  # Navy blue color\n\n# Calculate the mean, max, min, and count of Health Score for each Occupation\nhealth_score_by_occupation = df_train.groupby('Occupation')['Health Score'].agg(['mean', 'max', 'min', 'count'])\n\n# Set up the figure and axes for subplots with a larger size\nfig, axes = plt.subplots(2, 2, figsize=(22, 18))  # Increased figsize for better prominence\n\n# Plot 1: Bar plot of mean Health Score by Occupation with the consistent color palette\nsns.barplot(x=health_score_by_occupation.index, \n            y=health_score_by_occupation['mean'], \n            ax=axes[0, 0], \n            palette=occupation_color_map)\naxes[0, 0].set_title('Mean Health Score by Occupation', fontsize=24, fontweight='bold')\naxes[0, 0].set_xlabel('Occupation', fontsize=18)\naxes[0, 0].set_ylabel('Mean Health Score', fontsize=18)\naxes[0, 0].tick_params(axis='x', rotation=45, labelsize=16)\n\n# Create custom legend for bar plot\nlegend_labels = [Line2D([0], [0], marker='o', color='w', markerfacecolor=color, markersize=12) \n                 for color in occupation_color_map.values()]\naxes[0, 0].legend(legend_labels, occupation_color_map.keys(), title='Occupation', loc='upper left', bbox_to_anchor=(1, 1), fontsize=16)\n\n# Plot 2: Box plot for Health Score distribution by Occupation\nsns.boxplot(x='Occupation', y='Health Score', data=df_train, ax=axes[0, 1], palette=occupation_color_map)\naxes[0, 1].set_title('Health Score Distribution by Occupation', fontsize=24, fontweight='bold')\naxes[0, 1].set_xlabel('Occupation', fontsize=18)\naxes[0, 1].set_ylabel('Health Score', fontsize=18)\naxes[0, 1].tick_params(axis='x', rotation=45, labelsize=16)\n\n# Add custom legend for the box plot\naxes[0, 1].legend(legend_labels, occupation_color_map.keys(), title='Occupation', loc='upper left', bbox_to_anchor=(1, 1), fontsize=16)\n\n# Plot 3: Count plot of Occupation (to show how many records per occupation)\nsns.countplot(x='Occupation', data=df_train, ax=axes[1, 0], palette=occupation_color_map)\naxes[1, 0].set_title('Count of Records by Occupation', fontsize=24, fontweight='bold')\naxes[1, 0].set_xlabel('Occupation', fontsize=18)\naxes[1, 0].set_ylabel('Count', fontsize=18)\naxes[1, 0].tick_params(axis='x', rotation=45, labelsize=16)\n\n# Add custom legend for the count plot\naxes[1, 0].legend(legend_labels, occupation_color_map.keys(), title='Occupation', loc='upper left', bbox_to_anchor=(1, 1), fontsize=16)\n\n# Plot 4: Grouped Histogram for Health Score distribution by Occupation\nsns.histplot(data=df_train, x='Health Score', hue='Occupation', multiple='stack', kde=True, ax=axes[1, 1], palette=occupation_color_map, bins=20)\naxes[1, 1].set_title('Grouped Health Score Distribution by Occupation (Histogram)', fontsize=24, fontweight='bold')\naxes[1, 1].set_xlabel('Health Score', fontsize=18)\naxes[1, 1].set_ylabel('Frequency', fontsize=18)\n\n# Add custom legend for the histogram\naxes[1, 1].legend(legend_labels, occupation_color_map.keys(), title='Occupation', loc='upper left', bbox_to_anchor=(1, 1), fontsize=16)\n\n# Adjust layout for better spacing\nplt.tight_layout(pad=6.0)  # Increased padding for better spacing\n\n# Show the plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:23.227625Z","iopub.execute_input":"2024-12-25T06:13:23.227942Z","iopub.status.idle":"2024-12-25T06:13:31.066813Z","shell.execute_reply.started":"2024-12-25T06:13:23.227917Z","shell.execute_reply":"2024-12-25T06:13:31.06594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find the most frequent Occupation for each Location\nmost_frequent_occupation_by_location = df_train.groupby('Location')['Occupation'].agg(lambda x: x.mode()[0])\n\n# Output the most frequent Occupation for each Location\nprint(\"Most Frequent Occupation for each Location:\\n\", most_frequent_occupation_by_location)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:35.468272Z","iopub.execute_input":"2024-12-25T06:13:35.468567Z","iopub.status.idle":"2024-12-25T06:13:35.627689Z","shell.execute_reply.started":"2024-12-25T06:13:35.468543Z","shell.execute_reply":"2024-12-25T06:13:35.626923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the distribution of Occupation for each Location\noccupation_by_location = df_train.groupby('Location')['Occupation'].value_counts(normalize=True).unstack().fillna(0)\n\n# Calculate the mean, max, and count of Health Score for each Location and Occupation\nhealth_score_by_location_occupation = df_train.groupby(['Location', 'Occupation'])['Health Score'].agg(['count', 'mean', 'max'])\n\n# Output the results\ndisplay(\"Occupation Distribution within each Location:\", occupation_by_location)\ndisplay(\"Health Score by Location and Occupation:\", health_score_by_location_occupation)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:36.489215Z","iopub.execute_input":"2024-12-25T06:13:36.489498Z","iopub.status.idle":"2024-12-25T06:13:36.8043Z","shell.execute_reply.started":"2024-12-25T06:13:36.489477Z","shell.execute_reply":"2024-12-25T06:13:36.803599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute missing values in 'Occupation' with the mode (most frequent value)\noccupation_imputer = SimpleImputer(strategy='most_frequent')\n\n# Ensure the input to fit_transform is a 2D array by selecting the column as DataFrame\ndf_train['Occupation'] = occupation_imputer.fit_transform(df_train[['Occupation']]).flatten()\ndf_test['Occupation'] = occupation_imputer.transform(df_test[['Occupation']]).flatten()\n\n# Define the custom color palette for each plot\noccupation_palette = ['#B77A00', '#001F2D', '#29002d']\n\n# Set up the figure and axes for subplots with much larger figsize\nfig, axes = plt.subplots(2, 2, figsize=(40, 32))  # Much larger figsize\n\n# Plot 1: Stacked bar plot of Occupation distribution within each Location\noccupation_by_location.plot(kind='bar', stacked=True, ax=axes[0, 0], color=occupation_palette)\naxes[0, 0].set_title('Occupation Distribution within Each Location', fontsize=30, fontweight='bold')\naxes[0, 0].set_xlabel('Location', fontsize=24)\naxes[0, 0].set_ylabel('Proportion', fontsize=24)\naxes[0, 0].tick_params(axis='x', rotation=45, labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[0, 0].tick_params(axis='y', labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[0, 0].legend(title='Occupation', fontsize=20, bbox_to_anchor=(1.05, 1), loc='upper left')  # Legend outside\n\n# Add bold values on bars for Plot 1\nfor p in axes[0, 0].patches:\n    height = p.get_height()\n    width = p.get_width()\n    x, y = p.get_xy()  # Get the x and y coordinates of the rectangle\n    axes[0, 0].text(x + width / 2, y + height / 2, f'{height:.2f}', ha='center', va='center', fontsize=18, color='white', fontweight='bold')\n\n# Plot 2: Boxplot of Health Score by Location and Occupation\nsns.boxplot(x='Location', y='Health Score', hue='Occupation', data=df_train, ax=axes[0, 1], palette=occupation_palette)\naxes[0, 1].set_title('Health Score Distribution by Location and Occupation (Boxplot)', fontsize=30, fontweight='bold')\naxes[0, 1].set_xlabel('Location', fontsize=24)\naxes[0, 1].set_ylabel('Health Score', fontsize=24)\naxes[0, 1].tick_params(axis='x', rotation=45, labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[0, 1].tick_params(axis='y', labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[0, 1].legend(title='Occupation', fontsize=20, bbox_to_anchor=(1.05, 1), loc='upper left')  # Legend outside\n\n# Plot 3: Clustered bar plot of Health Score count by Location and Occupation\nhealth_score_by_location_occupation['count'].unstack().plot(kind='bar', ax=axes[1, 0], color=occupation_palette, width=0.8)\naxes[1, 0].set_title('Health Score Count by Location and Occupation', fontsize=30, fontweight='bold')\naxes[1, 0].set_xlabel('Location', fontsize=24)\naxes[1, 0].set_ylabel('Count', fontsize=24)\naxes[1, 0].tick_params(axis='x', rotation=45, labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[1, 0].tick_params(axis='y', labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[1, 0].legend(title='Occupation', fontsize=20, bbox_to_anchor=(1.05, 1), loc='upper left')  # Legend outside\n\n# Add bold values on bars for Plot 3\nfor p in axes[1, 0].patches:\n    height = p.get_height()\n    width = p.get_width()\n    x, y = p.get_xy()  # Get the x and y coordinates of the rectangle\n    axes[1, 0].text(x + width / 2, y + height / 2, f'{height:.0f}', ha='center', va='center', fontsize=18, color='white', fontweight='bold')\n\n# Plot 4: Clustered bar plot of mean Health Score by Location and Occupation\nhealth_score_by_location_occupation['mean'].unstack().plot(kind='bar', ax=axes[1, 1], color=occupation_palette, width=0.8)\naxes[1, 1].set_title('Mean Health Score by Location and Occupation', fontsize=30, fontweight='bold')\naxes[1, 1].set_xlabel('Location', fontsize=24)\naxes[1, 1].set_ylabel('Mean Health Score', fontsize=24)\naxes[1, 1].tick_params(axis='x', rotation=45, labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[1, 1].tick_params(axis='y', labelsize=24, width=2, length=10, colors='black', grid_color='gray', grid_alpha=0.5)\naxes[1, 1].legend(title='Occupation', fontsize=20, bbox_to_anchor=(1.05, 1), loc='upper left')  # Legend outside\n\n# Add bold values on bars for Plot 4\nfor p in axes[1, 1].patches:\n    height = p.get_height()\n    width = p.get_width()\n    x, y = p.get_xy()  # Get the x and y coordinates of the rectangle\n    axes[1, 1].text(x + width / 2, y + height / 2, f'{height:.2f}', ha='center', va='center', fontsize=18, color='white', fontweight='bold')\n\n# Adjust layout for better spacing\nplt.tight_layout(pad=8.0)  # Increased padding for better spacing\n\n# Show the plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:37.198868Z","iopub.execute_input":"2024-12-25T06:13:37.199256Z","iopub.status.idle":"2024-12-25T06:13:39.676397Z","shell.execute_reply.started":"2024-12-25T06:13:37.199222Z","shell.execute_reply":"2024-12-25T06:13:39.675459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a threshold for low Health Score (e.g., Health Score < 20)\nlow_health_score_threshold = 20\n# Define a threshold for high Health Score (e.g., Health Score >= 50)\nhigh_health_score_threshold = 50\n# Calculate the proportion of low-health individuals (Health Score < threshold) by Location and Occupation\nlow_health_score_proportion = df_train[df_train['Health Score'] < low_health_score_threshold] \\\n    .groupby(['Location', 'Occupation']).size() / df_train.groupby(['Location', 'Occupation']).size()\n\n# Calculate the proportion of high-health individuals (Health Score >= threshold) by Location and Occupation\nhigh_health_score_proportion = df_train[df_train['Health Score'] >= high_health_score_threshold] \\\n    .groupby(['Location', 'Occupation']).size() / df_train.groupby(['Location', 'Occupation']).size()\n# Combine both low and high health score proportions into a single DataFrame for easy comparison\nhealth_score_comparison = pd.DataFrame({\n    'Low Health Score Proportion': low_health_score_proportion,\n    'High Health Score Proportion': high_health_score_proportion\n}).fillna(0)  # Fill missing values with 0 if there are any categories without low or high health individuals\n\n# Output the comparative analysis\ndisplay(\"Comparative Analysis of Low and High Health Score Proportions by Location and Occupation:\", health_score_comparison)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:43.445749Z","iopub.execute_input":"2024-12-25T06:13:43.446123Z","iopub.status.idle":"2024-12-25T06:13:43.875977Z","shell.execute_reply.started":"2024-12-25T06:13:43.44609Z","shell.execute_reply":"2024-12-25T06:13:43.874996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reshape the data for the bar plot\nhealth_score_comparison_reset = health_score_comparison.reset_index()\n\n# Aggregate proportions for the pie chart\ntotal_low = health_score_comparison['Low Health Score Proportion'].sum()\ntotal_high = health_score_comparison['High Health Score Proportion'].sum()\n\n# Data for the pie chart\npie_data = [total_low, total_high]\npie_labels = ['Low Health Score', 'High Health Score']\noccupation_palette = ['#B77A00', '#001F2D']\n\n# Create the subplots\nfig, axes = plt.subplots(1, 2, figsize=(25, 10))\n\n# Bar plot\nsns.barplot(\n    data=health_score_comparison_reset.melt(\n        id_vars=['Location', 'Occupation'],\n        value_vars=['Low Health Score Proportion', 'High Health Score Proportion']\n    ),\n    x='Location',\n    y='value',\n    hue='variable',\n    ci=None,\n    palette=occupation_palette,\n    ax=axes[0]\n)\naxes[0].set_title('Comparative Proportions of Low and High Health Scores by Location and Occupation', fontsize=18, fontweight='bold')\naxes[0].set_xlabel('Location', fontsize=16, fontweight='bold')\naxes[0].set_ylabel('Proportion', fontsize=16, fontweight='bold')\naxes[0].tick_params(axis='x', rotation=45, labelsize=14, labelcolor='black')\naxes[0].tick_params(axis='y', labelsize=14, labelcolor='black')\n\n# Move legend outside the plot\naxes[0].legend(\n    title='Health Score Category', \n    fontsize=14, \n    loc='upper left', \n    bbox_to_anchor=(1, 1),\n    title_fontsize=14\n)\n\n# Add values above bars\nfor p in axes[0].patches:\n    height = p.get_height()\n    if not pd.isna(height):\n        axes[0].text(\n            p.get_x() + p.get_width() / 2., \n            height + 0.01, \n            f'{height:.2f}', \n            ha='center', fontsize=12, color='black', fontweight='bold'\n        )\n\n# Pie chart\nwedges, texts, autotexts = axes[1].pie(\n    pie_data,\n    labels=pie_labels,\n    autopct='%1.1f%%',\n    startangle=90,\n    colors=occupation_palette,\n    textprops={'fontsize': 14, 'fontweight': 'bold'},\n    wedgeprops={'edgecolor': 'black', 'linewidth': 1.2}\n)\n\n# Format pie chart values\nfor autotext in autotexts:\n    autotext.set_color('white')\n    autotext.set_fontweight('bold')\n\naxes[1].set_title('Overall Proportions of Low and High Health Scores', fontsize=18, fontweight='bold')\n\n# Adjust layout\nplt.tight_layout()\nplt.subplots_adjust(right=0.8)  # Adjust to make space for the legend\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:50.261932Z","iopub.execute_input":"2024-12-25T06:13:50.262281Z","iopub.status.idle":"2024-12-25T06:13:50.772081Z","shell.execute_reply.started":"2024-12-25T06:13:50.262252Z","shell.execute_reply":"2024-12-25T06:13:50.771201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the mean values of Age, Health Score, and Annual Income for each Occupation\noccupation_health_income_age_mean = df_train.groupby('Occupation')[['Age', 'Health Score', 'Annual Income']].mean()\n\n# Output the mean values for Age, Health Score, and Annual Income for each Occupation\nprint(\"Mean values of Age, Health Score, and Annual Income for each Occupation:\\n\", occupation_health_income_age_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:57.297423Z","iopub.execute_input":"2024-12-25T06:13:57.297758Z","iopub.status.idle":"2024-12-25T06:13:57.391687Z","shell.execute_reply.started":"2024-12-25T06:13:57.297731Z","shell.execute_reply":"2024-12-25T06:13:57.390802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the custom color palette for each plot\noccupation_palette = ['#B77A00', '#001F2D', '#29002d']\n\n# Set a style for the plots\nsns.set_style(\"whitegrid\")\n\n# Create subplots\nfig, axes = plt.subplots(nrows=2, ncols=3, figsize=(24, 16))  # 2 rows and 3 columns\n\n# --- First Row: Bar Plots ---\n# Vertical Bar Plot for Age\nsns.barplot(\n    x=occupation_health_income_age_mean.index,  # 'Occupation' on the y-axis\n    y=occupation_health_income_age_mean['Age'],  # 'Mean Age' on the x-axis\n    color=occupation_palette[0],\n    ax=axes[0, 0]\n)\naxes[0, 0].set_title('Mean Age by Occupation', fontsize=18, fontweight='bold', color=occupation_palette[0])\naxes[0, 0].set_ylabel('Occupation', fontsize=16, fontweight='bold', color=occupation_palette[0])\naxes[0, 0].set_xlabel('Mean Age', fontsize=16, fontweight='bold', color=occupation_palette[0])\naxes[0, 0].tick_params(axis='x', labelsize=14, rotation=45, length=6)\naxes[0, 0].tick_params(axis='y', labelsize=14, length=6)\n\n# Annotating bars with values\nfor p in axes[0, 0].patches:\n    axes[0, 0].annotate(f'{p.get_height():.2f}', (p.get_x() + p.get_width() / 4., p.get_height()),\n                        ha='center', va='center', fontsize=14, fontweight='bold', color='black', xytext=(0, 10), textcoords='offset points')\n\n# Vertical Bar Plot for Health Score\nsns.barplot(\n    x=occupation_health_income_age_mean.index,  # 'Occupation' on the y-axis\n    y=occupation_health_income_age_mean['Health Score'],  # 'Mean Health Score' on the x-axis\n    color=occupation_palette[1],\n    ax=axes[0, 1]\n)\naxes[0, 1].set_title('Mean Health Score by Occupation', fontsize=18, fontweight='bold', color=occupation_palette[1])\naxes[0, 1].set_ylabel('Occupation', fontsize=16, fontweight='bold', color=occupation_palette[1])\naxes[0, 1].set_xlabel('Mean Health Score', fontsize=16, fontweight='bold', color=occupation_palette[1])\naxes[0, 1].tick_params(axis='x', labelsize=14, rotation=45, length=6)\naxes[0, 1].tick_params(axis='y', labelsize=14, length=6)\n\n# Annotating bars with values\nfor p in axes[0, 1].patches:\n    axes[0, 1].annotate(f'{p.get_height():.2f}', (p.get_x() + p.get_width() / 4., p.get_height()),\n                        ha='center', va='center', fontsize=14, fontweight='bold', color='black', xytext=(0, 10), textcoords='offset points')\n\n# Vertical Bar Plot for Annual Income\nsns.barplot(\n    x=occupation_health_income_age_mean.index,  # 'Occupation' on the y-axis\n    y=occupation_health_income_age_mean['Annual Income'],  # 'Mean Annual Income' on the x-axis\n    color=occupation_palette[2],\n    ax=axes[0, 2]\n)\naxes[0, 2].set_title('Mean Annual Income by Occupation', fontsize=18, fontweight='bold', color=occupation_palette[2])\naxes[0, 2].set_ylabel('Occupation', fontsize=16, fontweight='bold', color=occupation_palette[2])\naxes[0, 2].set_xlabel('Mean Annual Income', fontsize=16, fontweight='bold', color=occupation_palette[2])\naxes[0, 2].tick_params(axis='x', labelsize=14, rotation=45, length=6)\naxes[0, 2].tick_params(axis='y', labelsize=14, length=6)\n\n# Annotating bars with values\nfor p in axes[0, 2].patches:\n    axes[0, 2].annotate(f'{p.get_height():.2f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                        ha='center', va='center', fontsize=14, fontweight='bold', color='black', xytext=(0, 10), textcoords='offset points')\n\n# --- Second Row: Pie Charts ---\n# Pie Chart for Mean Age\nage_labels = occupation_health_income_age_mean.index\nage_sizes = occupation_health_income_age_mean['Age']\npie1 = axes[1, 0].pie(age_sizes, labels=age_labels, autopct='%1.1f%%', colors=occupation_palette, startangle=90, \n                      textprops={'fontsize': 14, 'fontweight': 'bold', 'color': 'white'}, wedgeprops={'edgecolor': 'black'})\naxes[1, 0].set_title('Mean Age Distribution by Occupation', fontsize=18, fontweight='bold', color=occupation_palette[0])\n\n# Add legend for Age\naxes[1, 0].legend(age_labels, loc='upper right', bbox_to_anchor=(1.3, 1), fontsize=14, title=\"Occupations\")\n\n# Pie Chart for Mean Health Score\nhealth_score_labels = occupation_health_income_age_mean.index\nhealth_score_sizes = occupation_health_income_age_mean['Health Score']\npie2 = axes[1, 1].pie(health_score_sizes, labels=health_score_labels, autopct='%1.1f%%', colors=occupation_palette, startangle=90, \n                      textprops={'fontsize': 14, 'fontweight': 'bold', 'color': 'white'}, wedgeprops={'edgecolor': 'black'})\naxes[1, 1].set_title('Mean Health Score Distribution by Occupation', fontsize=18, fontweight='bold', color=occupation_palette[1])\n\n# Add legend for Health Score\naxes[1, 1].legend(health_score_labels, loc='upper right', bbox_to_anchor=(1.3, 1), fontsize=14, title=\"Occupations\")\n\n# Pie Chart for Mean Annual Income\nincome_labels = occupation_health_income_age_mean.index\nincome_sizes = occupation_health_income_age_mean['Annual Income']\npie3 = axes[1, 2].pie(income_sizes, labels=income_labels, autopct='%1.1f%%', colors=occupation_palette, startangle=90, \n                      textprops={'fontsize': 14, 'fontweight': 'bold', 'color': 'white'}, wedgeprops={'edgecolor': 'black'})\naxes[1, 2].set_title('Mean Annual Income Distribution by Occupation', fontsize=18, fontweight='bold', color=occupation_palette[2])\n\n# Add legend for Annual Income\naxes[1, 2].legend(income_labels, loc='upper right', bbox_to_anchor=(1.3, 1), fontsize=14, title=\"Occupations\")\n\n# Adjust layout\nplt.tight_layout()\nplt.subplots_adjust(hspace=0.4, wspace=0.6)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:13:58.525087Z","iopub.execute_input":"2024-12-25T06:13:58.52541Z","iopub.status.idle":"2024-12-25T06:13:59.646888Z","shell.execute_reply.started":"2024-12-25T06:13:58.525383Z","shell.execute_reply":"2024-12-25T06:13:59.646109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n### 1. Health Score Statistics\n- **Maximum Mean Health Score**:\n  - People who are **Unemployed**: **25.681818**.\n- **Maximum Health Score**:\n  - People who are **Employed**: **58.886035**.\n\n### 2. Occupational Distribution\n- **Highest Count**:\n  - **Self-Employed**: **265,143**.\n- **Highest Frequency Count by Location**:\n  - **Rural Location**: **0.534684** (for people who are **Employed**).\n- **Lowest Frequency Count by Location**:\n  - **Rural Location**: **0.229896** (for people who are **Unemployed**).\n- **Intermediate Frequency Count by Location**:\n  - **Urban Location**: **0.236165** (for people who are **Self-Employed**).\n\n### 3. Mean Frequency Distribution\n- **Highest Mean Frequency Count**:\n  - **Unemployed in Rural Areas**: **25.737549**.\n- **Lowest Mean Frequency Count**:\n  - **Self-Employed in Urban Areas**: **25.486259**.\n\n### 4. Proportion of Occupation by Location\n- **Highest Proportion**:\n  - People who are **Employed** across rural, urban, and suburban locations: **0.53**.\n\n### 5. Health Scores by Location and Occupation\n- **Employed**:\n  - Urban Areas: **198,176**.\n  - Suburban Areas: **201,817**.\n  - Rural Areas: **201,037**.\n- **Unemployed**:\n  - Urban Areas: **85,953**.\n  - Suburban Areas: **86,861**.\n  - Rural Areas: **86,400**.\n\n### 6. Mean Health Scores by Location and Occupation\n- **Highest Mean Health Score**:\n  - People in **Rural Areas** who are **Unemployed**: **25.94**.\n- **Lowest Mean Health Score**:\n  - People in **Urban Areas** who are **Employed**: **25.49**.\n\n### 7. Health Score Proportions\n- **Highest Health Score Proportion**:\n  - People in **Suburban Areas** who are **Employed**: **0.025905**.\n- **Lowest Health Score Proportion**:\n  - People in **Rural Areas** who are **Unemployed**: **0.342692**.\n\n### 8. Overall Health Score Proportions\n- **Lowest Health Proportion**: **93.2%**.\n- **Highest Health Proportion**: **6.8%**.\n\n### 9. Mean Metrics by Occupation\n- **Employed**:\n  - Mean Age: **41.1219**.\n  - Mean Health Score: **25.5959**.\n  - Mean Annual Income: **32,621.53**.\n- **Self-Employed**:\n  - Mean Age: **41.1866**.\n  - Mean Health Score: **25.5884**.\n  - Mean Annual Income: **32,909.19**.\n- **Unemployed**:\n  - Mean Age: **41.1584**.\n  - Mean Health Score: **25.6818**.\n  - Mean Annual Income: **32,864.54**.\n\n### 10. Key Insights\n- **Highest Mean Age Distribution**:\n  - People who are **Self-Employed**: **33.4%**.\n- **Highest Health Score Proportion**:\n  - People who are **Unemployed**: **33.4%**.\n- **Maximum Annual Income**:\n  - People who are **Self-Employed**.\n\n---\n\nThis formatting ensures the information is concise, well-organized, and easy to read. You can copy and paste this into a Markdown cell in your Jupyter Notebook and render it by pressing `Shift + Enter`. Let me know if you need further assistance!\n","metadata":{}},{"cell_type":"code","source":"# Analyze the frequency of different Policy Types in each Location\npolicy_type_by_location = df_train.groupby('Policy Type')['Location'].value_counts().unstack().fillna(0)\n\n# Output the distribution of Policy Type for each Location\nprint(\"Distribution of Policy Type by Location:\\n\", policy_type_by_location)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:08.57897Z","iopub.execute_input":"2024-12-25T06:14:08.579297Z","iopub.status.idle":"2024-12-25T06:14:08.72609Z","shell.execute_reply.started":"2024-12-25T06:14:08.579272Z","shell.execute_reply":"2024-12-25T06:14:08.725382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set a style for the plot\nsns.set(style=\"whitegrid\")\n\n# Grouping the data by Policy Type and Location\npolicy_type_by_location = df_train.groupby('Policy Type')['Location'].value_counts().unstack().fillna(0)\n\n# Define the custom color palette for each Policy Type (use colors from gender_palette)\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']\n\n# Create a bar plot for the distribution of Policy Type by Location\nax = policy_type_by_location.plot(kind='bar', stacked=True, figsize=(18, 8), color=gender_palette)\n\n# Adding titles and labels\nax.set_title('Distribution of Policy Type by Location', fontsize=16, fontweight='bold')\nax.set_xlabel('Policy Type', fontsize=14)\nax.set_ylabel('Frequency', fontsize=14)\nax.tick_params(axis='x', rotation=45, labelsize=12)\nax.tick_params(axis='y', labelsize=12)\n\n# Display the value counts inside the bars with bold text\nfor p in ax.patches:\n    height = p.get_height()\n    width = p.get_width()\n    x = p.get_x() + width / 2\n    y = p.get_y() + height / 2  # Positioning the text inside the bar\n    ax.annotate(f'{height:.0f}', (x, y), ha='center', va='center', fontsize=10, color='white', fontweight='bold')\n\n# Show the plot\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:10.145515Z","iopub.execute_input":"2024-12-25T06:14:10.145806Z","iopub.status.idle":"2024-12-25T06:14:10.60647Z","shell.execute_reply.started":"2024-12-25T06:14:10.145782Z","shell.execute_reply":"2024-12-25T06:14:10.605554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'Policy Type' and calculate the mean values of 'Previous Claims', 'Credit Score', and 'Vehicle Age',\n# as well as the count of these columns\nclaims_by_policy_type = df_train.groupby('Policy Type').agg(\n    Previous_Claims_Mean=('Previous Claims', 'mean'),\n    Credit_Score_Mean=('Credit Score', 'mean'),\n    Vehicle_Age_Mean=('Vehicle Age', 'mean'),\n    Policy_Type_Count=('Policy Type', 'size'),\n    Previous_Claims_Count=('Previous Claims', 'count'),\n    Credit_Score_Count=('Credit Score', 'count'),\n    Vehicle_Age_Count=('Vehicle Age', 'count')\n)\n\n# Reset the index to make 'Policy Type' a column, and put all column names in one row\nclaims_by_policy_type_reset = claims_by_policy_type.reset_index()\n\n# Output the detailed summary by Policy Type\ndisplay(claims_by_policy_type_reset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:13.973096Z","iopub.execute_input":"2024-12-25T06:14:13.9734Z","iopub.status.idle":"2024-12-25T06:14:14.095966Z","shell.execute_reply.started":"2024-12-25T06:14:13.973377Z","shell.execute_reply":"2024-12-25T06:14:14.095234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set a style for the plot\nsns.set(style=\"whitegrid\")\n\n# Define the custom color palette for each metric (mean and count)\nmean_colors = ['#1f77b4', '#ff7f0e', '#2ca02c']  # Default mean colors\ncount_colors = ['#d62728', '#9467bd', '#8c564b']  # Default count colors\n\n# Custom color palette for 'Credit Score' (used for both mean and count)\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']\n\n# Create a figure with two subplots (one for mean and one for count)\nfig, axes = plt.subplots(1, 2, figsize=(18, 8))\n\n# Bar width and position adjustment\nbar_width = 0.25\nindex = np.arange(len(claims_by_policy_type_reset))\n\n# Plotting Mean Values in the first subplot (not stacked)\nmean_bars_1 = axes[0].bar(index, claims_by_policy_type_reset['Previous_Claims_Mean'], bar_width, color=mean_colors[0], label='Mean Previous Claims')\nmean_bars_2 = axes[0].bar(index + bar_width, claims_by_policy_type_reset['Credit_Score_Mean'], bar_width, color=gender_palette[0], label='Mean Credit Score')  # Apply custom color for Credit Score\nmean_bars_3 = axes[0].bar(index + 2 * bar_width, claims_by_policy_type_reset['Vehicle_Age_Mean'], bar_width, color=mean_colors[2], label='Mean Vehicle Age')\n\n# Adding values above the bars for Mean Values\nfor bar in mean_bars_1:\n    yval = bar.get_height()\n    axes[0].text(bar.get_x() + bar.get_width() / 2, yval + 0.05, f'{yval:.2f}', ha='center', va='bottom', fontsize=10)\nfor bar in mean_bars_2:\n    yval = bar.get_height()\n    axes[0].text(bar.get_x() + bar.get_width() / 2, yval + 0.05, f'{yval:.2f}', ha='center', va='bottom', fontsize=10)\nfor bar in mean_bars_3:\n    yval = bar.get_height()\n    axes[0].text(bar.get_x() + bar.get_width() / 2, yval + 0.05, f'{yval:.2f}', ha='center', va='bottom', fontsize=10)\n\n# Adding titles, labels, and legend for the first subplot\naxes[0].set_title('Mean Values by Policy Type', fontsize=16, fontweight='bold')\naxes[0].set_xlabel('Policy Type', fontsize=14)\naxes[0].set_ylabel('Mean Values', fontsize=14)\naxes[0].tick_params(axis='x', rotation=45, labelsize=12)\naxes[0].tick_params(axis='y', labelsize=12)\naxes[0].set_xticks(index + bar_width)\naxes[0].set_xticklabels(claims_by_policy_type_reset['Policy Type'], fontsize=12)\naxes[0].legend(title='Mean Metrics', loc='upper left', bbox_to_anchor=(1, 1))\n\n# Plotting Count Values in the second subplot (not stacked)\ncount_bars_1 = axes[1].bar(index, claims_by_policy_type_reset['Previous_Claims_Count'], bar_width, color=count_colors[0], label='Count Previous Claims')\ncount_bars_2 = axes[1].bar(index + bar_width, claims_by_policy_type_reset['Credit_Score_Count'], bar_width, color=gender_palette[1], label='Count Credit Score')  # Apply custom color for Credit Score\ncount_bars_3 = axes[1].bar(index + 2 * bar_width, claims_by_policy_type_reset['Vehicle_Age_Count'], bar_width, color=count_colors[2], label='Count Vehicle Age')\n\n# Adding values above the bars for Count Values\nfor bar in count_bars_1:\n    yval = bar.get_height()\n    axes[1].text(bar.get_x() + bar.get_width() / 2, yval + 0.05, f'{yval:.0f}', ha='center', va='bottom', fontsize=10)\nfor bar in count_bars_2:\n    yval = bar.get_height()\n    axes[1].text(bar.get_x() + bar.get_width() / 2, yval + 0.05, f'{yval:.0f}', ha='center', va='bottom', fontsize=10)\nfor bar in count_bars_3:\n    yval = bar.get_height()\n    axes[1].text(bar.get_x() + bar.get_width() / 2, yval + 0.05, f'{yval:.0f}', ha='center', va='bottom', fontsize=10)\n\n# Adding titles, labels, and legend for the second subplot\naxes[1].set_title('Count Values by Policy Type', fontsize=16, fontweight='bold')\naxes[1].set_xlabel('Policy Type', fontsize=14)\naxes[1].set_ylabel('Count Values', fontsize=14)\naxes[1].tick_params(axis='x', rotation=45, labelsize=12)\naxes[1].tick_params(axis='y', labelsize=12)\naxes[1].set_xticks(index + bar_width)\naxes[1].set_xticklabels(claims_by_policy_type_reset['Policy Type'], fontsize=12)\naxes[1].legend(title='Count Metrics', loc='upper left', bbox_to_anchor=(1, 1))\n\n# Adjust the layout to avoid overlapping\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:15.34552Z","iopub.execute_input":"2024-12-25T06:14:15.345819Z","iopub.status.idle":"2024-12-25T06:14:16.028736Z","shell.execute_reply.started":"2024-12-25T06:14:15.345796Z","shell.execute_reply":"2024-12-25T06:14:16.027802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create bins for Vehicle Age\nvehicle_age_bins = [0, 2, 5, 10, 20]\nvehicle_age_labels = ['0-2', '2-5', '5-10', '10-20']\n\n# Categorize Vehicle Age into bins\ndf_train['Vehicle Age Range'] = pd.cut(df_train['Vehicle Age'], bins=vehicle_age_bins, labels=vehicle_age_labels)\n\n# Analyze Previous Claims by Vehicle Age and Credit Score Range\nclaims_by_vehicle_age_credit = df_train.groupby(['Credit Score', 'Vehicle Age Range'])['Previous Claims'].mean().unstack()\n\n# Output the relationship between Vehicle Age, Credit Score, and Previous Claims\ndisplay(\"Previous Claims by Vehicle Age and Credit Score:\", claims_by_vehicle_age_credit.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:20.143874Z","iopub.execute_input":"2024-12-25T06:14:20.144219Z","iopub.status.idle":"2024-12-25T06:14:20.24389Z","shell.execute_reply.started":"2024-12-25T06:14:20.144191Z","shell.execute_reply":"2024-12-25T06:14:20.243228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create bins for Vehicle Age\nvehicle_age_bins = [0, 2, 5, 10, 20]\nvehicle_age_labels = ['0-2', '2-5', '5-10', '10-20']\n\n# Categorize Vehicle Age into bins\ndf_train['Vehicle Age Range'] = pd.cut(df_train['Vehicle Age'], bins=vehicle_age_bins, labels=vehicle_age_labels)\n\n# Analyze Previous Claims by Vehicle Age and Credit Score Range\nclaims_by_vehicle_age_credit = df_train.groupby(['Credit Score', 'Vehicle Age Range'])['Previous Claims'].mean().unstack()\n\n# Output the relationship between Vehicle Age, Credit Score, and Previous Claims\nclaims_top_20 = claims_by_vehicle_age_credit.head(20)\n\n# Reset the index so 'Credit Score' and 'Vehicle Age Range' become columns\nclaims_top_20 = claims_top_20.reset_index()\n\n# Create a custom color palette for 'Credit Score'\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']\n\n# Create a scatter plot for 'Previous Claims' vs 'Vehicle Age Range', colored by 'Credit Score'\nplt.figure(figsize=(12, 6))\nsns.scatterplot(data=claims_top_20.melt(id_vars=['Credit Score'], value_vars=claims_top_20.columns[1:]),\n                x='Vehicle Age Range', y='value', hue='Credit Score', palette=gender_palette, s=100, marker='o')\n\n# Customize the plot\nplt.title('Scatter Plot of Previous Claims by Vehicle Age Range and Credit Score', fontsize=16, fontweight='bold')\nplt.xlabel('Vehicle Age Range', fontsize=14)\nplt.ylabel('Average Previous Claims', fontsize=14)\nplt.xticks(rotation=45, fontsize=12)\nplt.yticks(fontsize=12)\n\n# Adjust legend position to be outside the plot\nplt.legend(title='Credit Score', fontsize=12, bbox_to_anchor=(1.05, 1), loc='upper left')\n\n# Show the plot\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:21.438348Z","iopub.execute_input":"2024-12-25T06:14:21.438657Z","iopub.status.idle":"2024-12-25T06:14:22.131104Z","shell.execute_reply.started":"2024-12-25T06:14:21.438632Z","shell.execute_reply":"2024-12-25T06:14:22.130202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n### 1. Policy Type and Location Distribution\n- **Maximum Count**:\n  - Vehicles with **Comprehensive Policy Type** in **Rural Locations**: **133,781**.\n- **Minimum Count**:\n  - Vehicles with **Basic Policy Type** in **Rural Locations**: **131,974**.\n\n### 2. Credit Score and Vehicle Age Statistics\n- **Highest Mean Credit Score**:\n  - Vehicles with **Comprehensive Policy Type**:\n    - **Mean Credit Score**: **593.125983**.\n    - **Previous Mean Claim**: **1.004620**.\n    - **Mean Vehicle Age**: **9.568042**.\n    - **Previous Claim Count**: **278,142**.\n- **Lowest Mean Credit Score**:\n  - Vehicles with **Premium Policy Type**:\n    - **Mean Credit Score**: **592.618163**.\n    - **Previous Mean Claim**: **0.998727**.\n    - **Mean Vehicle Age**: **9.578748**.\n    - **Previous Claim Count**: **279,634**.\n\n### 3. Mean Statistics by Policy Type\n- **Mean Previous Claims**:\n  - All Policy Types (**Basic, Comprehensive, Premium**): **1.0**.\n- **Mean Vehicle Age**:\n  - **Premium Policy Type**: **9.58** (highest).\n  - **Basic Policy Type**: **9.56** (lowest).\n- **Mean Credit Score**:\n  - **Comprehensive Policy Type**: **593.13** (highest).\n  - **Basic Policy Type**: **59.03** (lowest).\n\n### 4. Vehicle Age Count by Policy Type\n- **Highest Vehicle Age Count**:\n  - **Premium Policy Type**: **401,845**.\n- **Lowest Vehicle Age Count**:\n  - **Basic Policy Type**: **398,552**.\n\n### 5. Previous Claim Count by Policy Type\n- **Highest Previous Claim Count**:\n  - **Premium Policy Type**: **279,634**.\n- **Lowest Previous Claim Count**:\n  - **Basic Policy Type**: **278,142**.\n\n### 6. Credit Score Count by Policy Type\n- **Highest Credit Score Count**:\n  - **Premium Policy Type**: **355,628**.\n- **Lowest Credit Score Count**:\n  - **Basic Policy Type**: **352,965**.\n\n### 7. Most Frequent Data Insights\n- **Highest Credit Scores**:\n  - Vehicles with **Age Range 0–2** and **Previous Claims around 1.25**.\n- **Lowest Credit Scores**:\n  - Vehicles with **Age Range 5–10** and **Previous Claims around 0.75**.\n","metadata":{}},{"cell_type":"code","source":"# Convert 'Customer Feedback' to numeric, coercing errors to NaN\ndf_train['Customer Feedback'] = pd.to_numeric(df_train['Customer Feedback'], errors='coerce')\n\n# Calculate the average Insurance Duration by Policy Type\ninsurance_duration_by_policy = df_train.groupby('Policy Type')['Insurance Duration'].mean()\n\n# Calculate the total number of policies by Policy Type and Insurance Duration\npolicy_count_by_type_duration = df_train.groupby(['Policy Type', 'Insurance Duration']).size().unstack(fill_value=0)\n\n# Calculate the average Customer Feedback by Smoking Status\nfeedback_by_smoking_status = df_train.groupby('Smoking Status')['Customer Feedback'].mean()\n\n# Calculate the frequency of Smoking Status by Policy Type\nsmoking_status_by_policy = df_train.groupby('Policy Type')['Smoking Status'].value_counts().unstack(fill_value=0)\n\n# Merge all the results into a single DataFrame\npivot_table = pd.DataFrame({\n    'Average Insurance Duration': insurance_duration_by_policy\n})\n\n# Merge the total number of policies by Policy Type and Insurance Duration\npivot_table = pivot_table.join(policy_count_by_type_duration)\n\n# Merge the frequency of Smoking Status by Policy Type\npivot_table = pivot_table.join(smoking_status_by_policy)\n\n# Reset the index to make 'Policy Type' a column and put all column names in one row\npivot_table_reset = pivot_table.reset_index()\n\n# Output the combined pivot table\ndisplay(\"Combined Pivot Table:\", pivot_table_reset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:30.70774Z","iopub.execute_input":"2024-12-25T06:14:30.708092Z","iopub.status.idle":"2024-12-25T06:14:31.698147Z","shell.execute_reply.started":"2024-12-25T06:14:30.708063Z","shell.execute_reply":"2024-12-25T06:14:31.697262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'Policy Type' and calculate the mean values of 'Previous Claims', 'Credit Score', and 'Vehicle Age',\n# as well as the count of these columns\nclaims_by_policy_type = df_train.groupby('Policy Type').agg(\n    Previous_Claims_Mean=('Previous Claims', 'mean'),\n    Credit_Score_Mean=('Credit Score', 'mean'),\n    Vehicle_Age_Mean=('Vehicle Age', 'mean'),\n    Policy_Type_Count=('Policy Type', 'size'),\n    Previous_Claims_Count=('Previous Claims', 'count'),\n    Credit_Score_Count=('Credit Score', 'count'),\n    Vehicle_Age_Count=('Vehicle Age', 'count')\n)\n\n# Reset the index to make 'Policy Type' a column, and put all column names in one row\nclaims_by_policy_type_reset = claims_by_policy_type.reset_index()\n\n# Output the detailed summary by Policy Type\ndisplay(claims_by_policy_type_reset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:34.047048Z","iopub.execute_input":"2024-12-25T06:14:34.047363Z","iopub.status.idle":"2024-12-25T06:14:34.173153Z","shell.execute_reply.started":"2024-12-25T06:14:34.047337Z","shell.execute_reply":"2024-12-25T06:14:34.172367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\n# Define a custom color palette based on the colors you provided\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']\n\n# Create the plot with multiple subplots (2 rows, 2 columns)\nfig, axes = plt.subplots(2, 2, figsize=(20, 14), sharex=False)\n\n# Define the policy types to use for the legend\npolicy_types = claims_by_policy_type_reset['Policy Type'].unique()\n\n# Bar plot for the mean values of 'Previous Claims'\nsns.barplot(x='Policy Type', y='Previous_Claims_Mean', data=claims_by_policy_type_reset, \n            hue='Policy Type', palette=gender_palette, ax=axes[0, 0])\naxes[0, 0].set_title('Mean of Previous Claims by Policy Type', fontsize=18, fontweight='bold')\naxes[0, 0].set_ylabel('Mean Previous Claims', fontsize=14)\naxes[0, 0].tick_params(axis='x', rotation=45, labelsize=12)\naxes[0, 0].tick_params(axis='y', labelsize=12)\naxes[0, 0].tick_params(axis='x', direction='in', length=6)  # Add x-axis ticks\n\n# Add the values above the bars\nfor p in axes[0, 0].patches:\n    axes[0, 0].annotate(f'{p.get_height():.2f}', \n                        (p.get_x() + p.get_width() / 2., p.get_height()), \n                        ha='center', va='center', fontsize=12, \n                        xytext=(0, 8), textcoords='offset points')\n\n# Bar plot for the mean values of 'Credit Score'\nsns.barplot(x='Policy Type', y='Credit_Score_Mean', data=claims_by_policy_type_reset, \n            hue='Policy Type', palette=gender_palette, ax=axes[0, 1])\naxes[0, 1].set_title('Mean of Credit Score by Policy Type', fontsize=18, fontweight='bold')\naxes[0, 1].set_ylabel('Mean Credit Score', fontsize=14)\naxes[0, 1].tick_params(axis='x', rotation=45, labelsize=12)\naxes[0, 1].tick_params(axis='y', labelsize=12)\naxes[0, 1].tick_params(axis='x', direction='in', length=6)  # Add x-axis ticks\n\n# Add the values above the bars\nfor p in axes[0, 1].patches:\n    axes[0, 1].annotate(f'{p.get_height():.2f}', \n                        (p.get_x() + p.get_width() / 2., p.get_height()), \n                        ha='center', va='center', fontsize=12, \n                        xytext=(0, 8), textcoords='offset points')\n\n# Bar plot for the mean values of 'Vehicle Age'\nsns.barplot(x='Policy Type', y='Vehicle_Age_Mean', data=claims_by_policy_type_reset, \n            hue='Policy Type', palette=gender_palette, ax=axes[1, 0])\naxes[1, 0].set_title('Mean of Vehicle Age by Policy Type', fontsize=18, fontweight='bold')\naxes[1, 0].set_ylabel('Mean Vehicle Age', fontsize=14)\naxes[1, 0].tick_params(axis='x', rotation=45, labelsize=12)\naxes[1, 0].tick_params(axis='y', labelsize=12)\naxes[1, 0].tick_params(axis='x', direction='in', length=6)  # Add x-axis ticks\n\n# Add the values above the bars\nfor p in axes[1, 0].patches:\n    axes[1, 0].annotate(f'{p.get_height():.2f}', \n                        (p.get_x() + p.get_width() / 2., p.get_height()), \n                        ha='center', va='center', fontsize=12, \n                        xytext=(0, 8), textcoords='offset points')\n\n# Bar plot for the counts of 'Previous Claims'\nsns.barplot(x='Policy Type', y='Previous_Claims_Count', data=claims_by_policy_type_reset, \n            hue='Policy Type', palette=gender_palette, ax=axes[1, 1])\naxes[1, 1].set_title('Count of Previous Claims by Policy Type', fontsize=18, fontweight='bold')\naxes[1, 1].set_ylabel('Count of Previous Claims', fontsize=14)\naxes[1, 1].tick_params(axis='x', rotation=45, labelsize=12)\naxes[1, 1].tick_params(axis='y', labelsize=12)\naxes[1, 1].tick_params(axis='x', direction='in', length=6)  # Add x-axis ticks\n\n# Add the values above the bars\nfor p in axes[1, 1].patches:\n    axes[1, 1].annotate(f'{p.get_height():.0f}', \n                        (p.get_x() + p.get_width() / 2., p.get_height()), \n                        ha='center', va='center', fontsize=12, \n                        xytext=(0, 8), textcoords='offset points')\n\n# Set common x-axis label for all subplots\nfig.text(0.5, 0.04, 'Policy Type', ha='center', fontsize=16, fontweight='bold')\n\n# Move the legend outside the plot to the right\nfor ax in axes.flat:\n    ax.legend(title='Policy Type', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# Adjust layout for better spacing\nplt.tight_layout(pad=4.0)\n\n# Display all the plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:35.180224Z","iopub.execute_input":"2024-12-25T06:14:35.180549Z","iopub.status.idle":"2024-12-25T06:14:36.304565Z","shell.execute_reply.started":"2024-12-25T06:14:35.180522Z","shell.execute_reply":"2024-12-25T06:14:36.303735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'Policy Type' and calculate the average 'Insurance Duration'\ninsurance_duration_by_policy = df_train.groupby('Policy Type')['Insurance Duration'].mean()\n\n# Output the average 'Insurance Duration' by Policy Type\ndisplay(\"Average Insurance Duration by Policy Type:\", insurance_duration_by_policy)\n\n# Group by 'Smoking Status' to calculate the average 'Customer Feedback'\nfeedback_by_smoking_status = df_train.groupby('Smoking Status')['Customer Feedback'].mean()\nprint(\"=============================================================================\")\n# Calculate the count of 'Smoking Status' for each 'Policy Type'\nsmoking_status_by_policy = df_train.groupby('Policy Type')['Smoking Status'].value_counts().unstack(fill_value=0)\n\n# Output the frequency of Smoking Status by 'Policy Type'\ndisplay(\"Frequency of Smoking Status by Policy Type:\", smoking_status_by_policy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:40.08316Z","iopub.execute_input":"2024-12-25T06:14:40.083463Z","iopub.status.idle":"2024-12-25T06:14:40.364047Z","shell.execute_reply.started":"2024-12-25T06:14:40.083439Z","shell.execute_reply":"2024-12-25T06:14:40.36334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define custom color palette\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']\n\n# Group by 'Policy Type' and calculate the average 'Insurance Duration'\ninsurance_duration_by_policy = df_train.groupby('Policy Type')['Insurance Duration'].mean()\n\n# Group by 'Smoking Status' to calculate the average 'Customer Feedback'\nfeedback_by_smoking_status = df_train.groupby('Smoking Status')['Customer Feedback'].mean()\n\n# Group by 'Policy Type' to calculate the count of 'Smoking Status'\nsmoking_status_by_policy = df_train.groupby('Policy Type')['Smoking Status'].value_counts().unstack(fill_value=0)\n\n# Create a subplot with 1 row and 2 columns\nfig, axes = plt.subplots(1, 2, figsize=(15, 7))\n\n# Pie Chart for the distribution of 'Insurance Duration' by 'Policy Type'\nwedges, texts, autotexts = axes[0].pie(\n    insurance_duration_by_policy, \n    labels=insurance_duration_by_policy.index, \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=gender_palette[:len(insurance_duration_by_policy)],\n    textprops={'color': 'white', 'fontweight': 'bold'},  # Text color and bold inside pie chart\n    wedgeprops={'edgecolor': 'black'},\n    labeldistance=1.15  # Move labels outside\n)\n\n# Make the pie chart labels bold\nfor autotext in autotexts:\n    autotext.set_fontweight('bold')\n\n# Change label color to black\nfor text in texts:\n    text.set_color('black')\n\n# Set pie chart title and add labels\naxes[0].set_title('Average Insurance Duration by Policy Type', fontsize=14, color='darkblue')\n\n# Add legend to pie chart\naxes[0].legend(wedges, insurance_duration_by_policy.index, title=\"Policy Type\", loc=\"center left\", bbox_to_anchor=(1, 0, 0.5, 1), fontsize=12)\n\n# Bar Chart for 'Smoking Status' frequency by 'Policy Type'\nsmoking_status_by_policy.plot(kind='bar', stacked=True, ax=axes[1], color=gender_palette, width=0.8)\n\n# Add the values inside the stacked bars\nfor p in axes[1].patches:\n    height = p.get_height()\n    width = p.get_width()\n    x, y = p.get_xy()  # Get the x and y position of the rectangle\n    axes[1].text(x + width/2, y + height/2, f'{int(height)}', ha='center', va='center', color='white', fontsize=12, fontweight='bold')\n\n# Add title, labels, and other enhancements to the bar chart\naxes[1].set_title('Frequency of Smoking Status by Policy Type', fontsize=14, color='darkgreen')\naxes[1].set_ylabel('Count of Smoking Status')\naxes[1].set_xlabel('Policy Type')\n\n# Improve layout and display\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:41.153391Z","iopub.execute_input":"2024-12-25T06:14:41.153701Z","iopub.status.idle":"2024-12-25T06:14:41.913296Z","shell.execute_reply.started":"2024-12-25T06:14:41.153677Z","shell.execute_reply":"2024-12-25T06:14:41.912424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by 'Policy Type' and 'Insurance Duration' to calculate the total number of policies\npolicy_count_by_type_duration = df_train.groupby(['Policy Type', 'Insurance Duration']).size().unstack(fill_value=0)\n\n# Output the number of policies by 'Policy Type' and 'Insurance Duration'\ndisplay(\"Number of Policies by Policy Type and Insurance Duration:\", policy_count_by_type_duration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:46.889571Z","iopub.execute_input":"2024-12-25T06:14:46.889919Z","iopub.status.idle":"2024-12-25T06:14:46.997895Z","shell.execute_reply.started":"2024-12-25T06:14:46.889888Z","shell.execute_reply":"2024-12-25T06:14:46.997079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a custom color palette\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712', '#34b1eb', '#eb3434', '#61eb34', '#776f7a', '#ff7d03']\n\n# Group by 'Policy Type' and 'Insurance Duration' to calculate the total number of policies\npolicy_count_by_type_duration = df_train.groupby(['Policy Type', 'Insurance Duration']).size().unstack(fill_value=0)\n\n# Create a subplot with 1 row and 1 column for the bar chart\nfig, ax = plt.subplots(figsize=(10, 6))\n\n# Stacked Bar Chart for Policy Count by Policy Type and Insurance Duration\n# Apply the custom color palette to the stacked bars\npolicy_count_by_type_duration.plot(kind='bar', stacked=True, ax=ax, width=0.8, color=gender_palette[:len(policy_count_by_type_duration.columns)])\n\n# Set title and labels for the chart\nax.set_title('Stacked Bar Chart: Policies by Type and Duration', fontsize=14, color='darkgreen')\nax.set_ylabel('Number of Policies', fontsize=12)\nax.set_xlabel('Policy Type', fontsize=12)\n\n# Add custom labels for Policy Type (ensure the index of the grouped data has correct Policy Types)\nax.set_xticklabels(policy_count_by_type_duration.index, rotation=45, ha='right', fontsize=10)\n\n# Add legend outside of the plot\nax.legend(title='Insurance Duration', loc='center left', bbox_to_anchor=(1, 0.5), fontsize=10)\n\n# Improve layout and display\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:47.441541Z","iopub.execute_input":"2024-12-25T06:14:47.441829Z","iopub.status.idle":"2024-12-25T06:14:47.994979Z","shell.execute_reply.started":"2024-12-25T06:14:47.441807Z","shell.execute_reply":"2024-12-25T06:14:47.994093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert 'Policy Start Date' to datetime format (if not already in datetime)\ndf_train['Policy Start Date'] = pd.to_datetime(df_train['Policy Start Date'], errors='coerce')\n\n# Extract the year and month from the Policy Start Date for analysis\ndf_train['Policy Start Year'] = df_train['Policy Start Date'].dt.year\ndf_train['Policy Start Month'] = df_train['Policy Start Date'].dt.month\n\n# Group by Policy Start Year and Month to calculate the average Insurance Duration\ninsurance_duration_by_start_date = df_train.groupby(['Policy Start Year', 'Policy Start Month'])['Insurance Duration'].mean()\n\n# Get the top 20 highest and lowest values\ntop_20_insurance_duration = insurance_duration_by_start_date.nlargest(20)\nbottom_20_insurance_duration = insurance_duration_by_start_date.nsmallest(20)\n\n# Output the top 20 highest and bottom 20 lowest average Insurance Duration by Policy Start Date\nprint(\"Highest Average Insurance Duration by Policy Start Date:\\n\", top_20_insurance_duration)\nprint(\"Lowest Average Insurance Duration by Policy Start Date:\\n\", bottom_20_insurance_duration)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:51.283554Z","iopub.execute_input":"2024-12-25T06:14:51.283868Z","iopub.status.idle":"2024-12-25T06:14:51.797785Z","shell.execute_reply.started":"2024-12-25T06:14:51.283845Z","shell.execute_reply":"2024-12-25T06:14:51.797104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\n# Combine the top 20 and bottom 20 values into a single DataFrame for easier plotting\ncombined_insurance_duration = pd.concat([top_20_insurance_duration, bottom_20_insurance_duration])\n\n# Reset index for easier plotting and manipulation\ncombined_insurance_duration = combined_insurance_duration.reset_index()\n\n# Define colors for the top and bottom groups\ncolor_palette = gender_palette[:2]  # Use the first two colors from the palette for the top and bottom\ntop_color = color_palette[0]  # Color for top 20 highest\nbottom_color = color_palette[1]  # Color for bottom 20 lowest\n\n# Create a new column to categorize the data as 'Top 20' or 'Bottom 20'\ncombined_insurance_duration['Category'] = ['Top 20'] * len(top_20_insurance_duration) + ['Bottom 20'] * len(bottom_20_insurance_duration)\n\n# Create the barplot\nplt.figure(figsize=(14, 7))\n\n# Plot the bars for the top 20 and bottom 20\nax = sns.barplot(x='Policy Start Year', y='Insurance Duration', hue='Category', data=combined_insurance_duration, dodge=True, palette=[top_color, bottom_color])\n\n# Adding titles, labels, and customizing the plot\nplt.title('Comparative Analysis of Insurance Duration by Policy Start Date', fontsize=16, fontweight='bold')\nplt.xlabel('Policy Start Year and Month', fontsize=14)\nplt.ylabel('Average Insurance Duration', fontsize=14)\nplt.xticks(rotation=45, fontsize=12)\nplt.yticks(fontsize=12)\n\n# Customizing the legend labels\nhandles, labels = plt.gca().get_legend_handles_labels()\nlabels = ['Highest Average Insurance Duration by Policy Start Date', 'Lowest Average Insurance Duration by Policy Start Date']\nplt.legend(handles=handles, labels=labels, title='Category', loc='upper left', bbox_to_anchor=(1, 1), fontsize=12)\n\n# Add values above the bars (including zero values)\nfor p in ax.patches:\n    height = p.get_height()\n    ax.annotate(f'{height:.2f}', \n                (p.get_x() + p.get_width() / 2., height), \n                ha='center', va='center', fontsize=12, color='black', \n                xytext=(0, 5), textcoords='offset points')\n\n# Display the plot\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:51.798938Z","iopub.execute_input":"2024-12-25T06:14:51.799285Z","iopub.status.idle":"2024-12-25T06:14:52.268939Z","shell.execute_reply.started":"2024-12-25T06:14:51.799249Z","shell.execute_reply":"2024-12-25T06:14:52.268119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define Insurance Duration bins\ninsurance_duration_bins = [0, 5, 10]\ninsurance_duration_labels = ['0-5', '5-10']\n\n# Categorize Insurance Duration into bins\ndf_train['Insurance Duration Category'] = pd.cut(df_train['Insurance Duration'], bins=insurance_duration_bins, labels=insurance_duration_labels)\n\n# Calculate the frequency of Smoking Status within each Insurance Duration category\nsmoking_status_by_duration_category = df_train.groupby(['Insurance Duration Category', 'Smoking Status']).size().unstack(fill_value=0)\n\n# Output the frequency of Smoking Status by Insurance Duration category\nprint(\"Frequency of Smoking Status by Insurance Duration Category:\\n\", smoking_status_by_duration_category)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:56.372356Z","iopub.execute_input":"2024-12-25T06:14:56.372689Z","iopub.status.idle":"2024-12-25T06:14:56.477286Z","shell.execute_reply.started":"2024-12-25T06:14:56.372659Z","shell.execute_reply":"2024-12-25T06:14:56.476324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set the style for the plots\nsns.set(style=\"whitegrid\")\n\n# Define the color palette for Smoking Status (using similar shades as before)\nsmoking_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712']\n\n# Create the plot figure and axes\nfig, axes = plt.subplots(1, 2, figsize=(18, 8), sharey=True)\n\n# Plot for Smoking Status in each Insurance Duration category\nsmoking_status_by_duration_category.plot(kind='bar', stacked=True, ax=axes[0], color=smoking_palette)\n\n# Adding titles, labels, and legends for the first subplot\naxes[0].set_title('Smoking Status by Insurance Duration (0-5 years)', fontsize=16, fontweight='bold')\naxes[0].set_xlabel('Insurance Duration Category', fontsize=14)\naxes[0].set_ylabel('Frequency', fontsize=14)\naxes[0].tick_params(axis='x', rotation=45, labelsize=12)\naxes[0].tick_params(axis='y', labelsize=12)\n\n# Adding values inside the bars (bold text)\nfor p in axes[0].patches:\n    height = p.get_height()\n    width = p.get_width()\n    x = p.get_x() + width / 2\n    y = p.get_y() + height / 2  # Positioning the text inside the bar\n    axes[0].annotate(f'{height:.0f}', (x, y), ha='center', va='center', fontsize=10, color='white', fontweight='bold')\n\n# Create the second subplot with Smoking Status by Insurance Duration (5-10 years)\nsmoking_status_by_duration_category.plot(kind='bar', stacked=True, ax=axes[1], color=smoking_palette)\n\n# Adding titles, labels, and legends for the second subplot\naxes[1].set_title('Smoking Status by Insurance Duration (5-10 years)', fontsize=16, fontweight='bold')\naxes[1].set_xlabel('Insurance Duration Category', fontsize=14)\naxes[1].set_ylabel('Frequency', fontsize=14)\naxes[1].tick_params(axis='x', rotation=45, labelsize=12)\naxes[1].tick_params(axis='y', labelsize=12)\n\n# Adding values inside the bars (bold text)\nfor p in axes[1].patches:\n    height = p.get_height()\n    width = p.get_width()\n    x = p.get_x() + width / 2\n    y = p.get_y() + height / 2  # Positioning the text inside the bar\n    axes[1].annotate(f'{height:.0f}', (x, y), ha='center', va='center', fontsize=10, color='white', fontweight='bold')\n\n# Adjust the layout for better presentation\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:14:56.733316Z","iopub.execute_input":"2024-12-25T06:14:56.733585Z","iopub.status.idle":"2024-12-25T06:14:57.319999Z","shell.execute_reply.started":"2024-12-25T06:14:56.733563Z","shell.execute_reply":"2024-12-25T06:14:57.319229Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n### 1. Insurance Duration by Policy Type\n- **Highest Average Insurance Duration**:\n  - **Basic Policy Type**: **5.022572**.\n- **Comprehensive Policy Type**: **5.015766**.\n- **Premium Policy Type**: **5.016342**.\n- **Insurance Duration by Year**:\n  - Highest Average Duration: **2019**.\n  - **Years 2020, 2021, 2024** also exhibit high durations by policy.\n\n### 2. Smoking Status and Policy Type\n- **Basic Policy Type**:\n  - Vehicles with **Smoking Status \"Yes\"**: **200,493**.\n  - Vehicles with **No Smoking Status**: **198,061**.\n- **Premium Policy Type**:\n  - Vehicles with **Smoking Status \"No\"**: **200,557**.\n- **Frequency of Smoking Status by Insurance Duration**:\n  - **Duration Range 0–5**:\n    - Smoking Status **\"Yes\"**: **332,265**.\n  - **Duration Range 5–10**:\n    - Smoking Status **\"Yes\"**: **269,607**.\n\n### 3. Customer Feedback by Policy Type\n- **Premium Policy Type**:\n  - Average Insurance Duration: **5.016342**.\n  - **Highest Customer Feedback**:\n    - Feedback **1 (Good Feedback)**: **45,188**.\n- **Basic Policy Type**:\n  - Average Insurance Duration: **5.022572**.\n  - **Lowest Customer Feedback**:\n    - Feedback **3**: **43,502**.\n\n### 4. Credit Score by Policy Type\n- **Highest Average Credit Score**:\n  - **Comprehensive Policy Type**: **593.13**.\n- **Lowest Average Credit Score**:\n  - **Basic Policy Type**: **593.13**.\n\n### 5. Previous Claims by Policy Type\n- **Highest Previous Claim Count**:\n  - **Basic Policy Type**: **278,195**.\n\n### 6. Insurance Duration and Smoking Status by Count\n- **Highest Count for Insurance Duration 9**:\n  - **Basic Policy Type**: **46,099**.\n- **Highest Count for Insurance Duration 3**:\n  - **Basic Policy Type**: **43,502**.\n\n### 7. Summary of Insights\n- Vehicles with **Basic Policy Type** and **Smoking Status \"Yes\"** dominate counts for Insurance Duration ranges **0–5** and **5–10**.\n- **Customer Feedback** indicates the **Premium Policy Type** receives the most positive feedback, while **Basic Policy Type** has the least favorable feedback.\n- The **year 2019** shows the highest Insurance Duration average, with subsequent years (2020, 2021, 2024) maintaining notable durations across policies.\n","metadata":{}},{"cell_type":"code","source":"# Analysis of 'Exercise Frequency' column\nexercise_frequency = df_train['Exercise Frequency']\n\n# 1. Frequency Distribution of Exercise Frequency\nexercise_frequency_counts = exercise_frequency.value_counts(normalize=True).sort_index()\n\n# 2. Mean Premium Amount by Exercise Frequency\npremium_by_exercise = df_train.groupby('Exercise Frequency')['Premium Amount'].mean()\n\n# 3. Mean Age by Exercise Frequency\nage_by_exercise = df_train.groupby('Exercise Frequency')['Age'].mean()\n\n# 4. Total Premium Amount by Exercise Frequency\ntotal_premium_by_exercise = df_train.groupby('Exercise Frequency')['Premium Amount'].sum()\n\n# 5. Maximum Age by Exercise Frequency\nmax_age_by_exercise = df_train.groupby('Exercise Frequency')['Age'].max()\n\n# Combine the results into one DataFrame\ncombined_exercise_analysis = pd.DataFrame({\n    'Exercise Frequency Count': exercise_frequency_counts,\n    'Mean Premium Amount': premium_by_exercise,\n    'Mean Age': age_by_exercise,\n    'Total Premium Amount': total_premium_by_exercise,\n    'Max Age': max_age_by_exercise\n})\n\n# Reset the index to have column names as rows\ncombined_exercise_analysis = combined_exercise_analysis.reset_index()\n\n# Display the combined results\ndisplay(\"Exercise Frequency Analysis:\", combined_exercise_analysis)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:03.888609Z","iopub.execute_input":"2024-12-25T06:15:03.888895Z","iopub.status.idle":"2024-12-25T06:15:04.239215Z","shell.execute_reply.started":"2024-12-25T06:15:03.888872Z","shell.execute_reply":"2024-12-25T06:15:04.238501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a custom color palette\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712', '#34b1eb', '#eb3434', '#61eb34', '#776f7a', '#ff7d03']\n\n# Get the unique exercise frequencies and create a color palette based on the custom palette\nexercise_frequencies = df_train['Exercise Frequency'].unique()\ncolors = gender_palette[:len(exercise_frequencies)]  # Adjust the number of colors if needed\n\n# Create a dictionary to map Exercise Frequency to a specific color\ncolor_dict = dict(zip(exercise_frequencies, colors))\n\n# Create subplots with 3 rows and 2 columns for different plots\nfig, axes = plt.subplots(3, 2, figsize=(18, 16))  # Increased figsize for larger plots\n\n# 1. Bar Chart for Exercise Frequency Count\nfor i, freq in enumerate(exercise_frequencies):\n    axes[0, 0].bar(freq, combined_exercise_analysis[combined_exercise_analysis['Exercise Frequency'] == freq]['Exercise Frequency Count'].values[0],\n                   color=color_dict[freq], label=freq)\naxes[0, 0].set_title('Exercise Frequency Distribution', fontsize=16, color='darkblue')\naxes[0, 0].set_ylabel('Frequency', fontsize=12)\naxes[0, 0].set_xlabel('Exercise Frequency', fontsize=12)\naxes[0, 0].legend(title='Exercise Frequency', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# 2. Box Plot for Premium Amount by Exercise Frequency\nsns.boxplot(x='Exercise Frequency', y='Premium Amount', data=df_train, ax=axes[0, 1], palette=color_dict)\naxes[0, 1].set_title('Premium Amount by Exercise Frequency', fontsize=16, color='darkgreen')\naxes[0, 1].set_ylabel('Premium Amount', fontsize=12)\naxes[0, 1].set_xlabel('Exercise Frequency', fontsize=12)\naxes[0, 1].legend(title='Exercise Frequency', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# 3. Bar Chart for Mean Age by Exercise Frequency\nfor i, freq in enumerate(exercise_frequencies):\n    axes[1, 0].bar(freq, combined_exercise_analysis[combined_exercise_analysis['Exercise Frequency'] == freq]['Mean Age'].values[0],\n                   color=color_dict[freq], label=freq)\naxes[1, 0].set_title('Mean Age by Exercise Frequency', fontsize=16, color='purple')\naxes[1, 0].set_ylabel('Mean Age', fontsize=12)\naxes[1, 0].set_xlabel('Exercise Frequency', fontsize=12)\naxes[1, 0].legend(title='Exercise Frequency', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# 4. Stacked Bar Chart for Total Premium Amount by Exercise Frequency\nfor i, freq in enumerate(exercise_frequencies):\n    axes[1, 1].bar(freq, combined_exercise_analysis[combined_exercise_analysis['Exercise Frequency'] == freq]['Total Premium Amount'].values[0],\n                   color=color_dict[freq], label=freq)\naxes[1, 1].set_title('Total Premium Amount by Exercise Frequency', fontsize=16, color='red')\naxes[1, 1].set_ylabel('Total Premium Amount', fontsize=12)\naxes[1, 1].set_xlabel('Exercise Frequency', fontsize=12)\naxes[1, 1].legend(title='Exercise Frequency', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# Format y-axis of the 'Total Premium Amount' plot for short form\ndef format_currency(x, pos):\n    \"\"\"Function to format y-axis labels in short form (K, M, etc.)\"\"\"\n    if x >= 1e6:\n        return f'{x*1e-6:.1f}M'  # Millions\n    elif x >= 1e3:\n        return f'{x*1e-3:.1f}K'  # Thousands\n    else:\n        return f'{x:.0f}'\n\naxes[1, 1].yaxis.set_major_formatter(FuncFormatter(format_currency))\n\n# 5. Histogram for Age Distribution by Exercise Frequency (Grouped Mode)\nfor i, freq in enumerate(exercise_frequencies):\n    filtered_data = df_train[df_train['Exercise Frequency'] == freq]['Age']\n    axes[2, 0].hist(filtered_data, bins=10, alpha=0.7, color=color_dict[freq], label=freq, histtype='barstacked')\naxes[2, 0].set_title('Age Distribution by Exercise Frequency', fontsize=16, color='orange')\naxes[2, 0].set_ylabel('Count', fontsize=12)\naxes[2, 0].set_xlabel('Age', fontsize=12)\naxes[2, 0].legend(title='Exercise Frequency', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# Hide the empty subplot at axes[2, 1]\naxes[2, 1].axis('off')\n\n# Improve layout and display\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:06.226882Z","iopub.execute_input":"2024-12-25T06:15:06.227279Z","iopub.status.idle":"2024-12-25T06:15:08.629871Z","shell.execute_reply.started":"2024-12-25T06:15:06.227253Z","shell.execute_reply":"2024-12-25T06:15:08.629073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Bin Age into different categories\nage_bins = [18, 22, 25, 30, 35, 40, 45, 50, 55, 60, 64]\nage_labels = ['18-22', '23-25', '26-30', '31-35', '36-40', '41-45', '46-50', '51-55', '56-60', '61-64']\ndf_train['Age Group'] = pd.cut(df_train['Age'], bins=age_bins, labels=age_labels)\n\n# 2. Analyze mean Premium Amount by Age Group\npremium_by_age_group = df_train.groupby('Age Group')['Premium Amount'].mean().reset_index(name=\"Premium Amount\")\nprint(\"\\nMean Premium Amount by Age Group:\")\nprint(premium_by_age_group)\n\n# 3. Analyze the distribution of Premium Amount within each Age Group (only mean, count, max, and min)\npremium_age_group_desc = df_train.groupby('Age Group')['Premium Amount'].agg(['mean', 'count', 'max', 'min'])\nprint(\"\\nPremium Amount Distribution within Age Groups:\")\nprint(premium_age_group_desc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:11.869877Z","iopub.execute_input":"2024-12-25T06:15:11.87024Z","iopub.status.idle":"2024-12-25T06:15:11.950964Z","shell.execute_reply.started":"2024-12-25T06:15:11.870211Z","shell.execute_reply":"2024-12-25T06:15:11.950097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Bin Age into different categories\nage_bins = [18, 22, 25, 30, 35, 40, 45, 50, 55, 60, 64]\nage_labels = ['18-22', '23-25', '26-30', '31-35', '36-40', '41-45', '46-50', '51-55', '56-60', '61-64']\ndf_train['Age Group'] = pd.cut(df_train['Age'], bins=age_bins, labels=age_labels)\n\n# 2. Impute missing values for Age and Premium Amount\ndf_train['Age'] = df_train['Age'].fillna(df_train['Age'].median())\ndf_train['Premium Amount'] = df_train['Premium Amount'].fillna(df_train['Premium Amount'].median())\n\n# 3. Analyze mean Premium Amount by Age Group\npremium_by_age_group = df_train.groupby('Age Group')['Premium Amount'].mean().reset_index(name=\"Premium Amount\")\n\n# 4. Analyze the distribution of Premium Amount within each Age Group (mean, count, max, and min)\npremium_age_group_desc = df_train.groupby('Age Group')['Premium Amount'].agg(['mean', 'count', 'max', 'min'])\n\n# 5. Create a single scatter plot for Premium Amount by Age Group\nplt.figure(figsize=(10, 6))\n\n# Define a custom color palette for gender\ngender_palette = ['#B77A00', '#03ecfc', '#fc030f', '#29002d', '#b85712', '#34b1eb', '#eb3434', '#61eb34', '#776f7a', '#ff7d03']\n\n# Plot the data for each age group using gender_palette\nfor i, age_group in enumerate(age_labels):\n    group_data = df_train[df_train['Age Group'] == age_group]\n    plt.scatter(group_data['Age'], group_data['Premium Amount'], \n                label=age_group, \n                color=gender_palette[i % len(gender_palette)],  # Loop over the color palette\n                alpha=0.7, edgecolors='w', s=50)\n\n# Set plot title and labels\nplt.title('Premium Amount vs Age Group', fontsize=16)\nplt.xlabel('Age', fontsize=12)\nplt.ylabel('Premium Amount', fontsize=12)\n\n# Adjust x-axis ticks to have a distance of 4\nplt.xticks(range(18, 65, 4), fontsize=12)  # Step size of 4 for the x-axis ticks\n\n# Add legend with the age group labels and corresponding colors outside the plot\nplt.legend(title='Age Group', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12)\n\n# Display the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:14.327948Z","iopub.execute_input":"2024-12-25T06:15:14.328301Z","iopub.status.idle":"2024-12-25T06:15:18.40867Z","shell.execute_reply.started":"2024-12-25T06:15:14.328276Z","shell.execute_reply":"2024-12-25T06:15:18.407725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by Gender and Age Group and calculate the mean Premium Amount\npremium_by_age_gender = df_train.groupby(['Gender', 'Age Group'])['Premium Amount'].mean().reset_index(name=\"Premium Amount\")\n\n# Sort the results by 'Premium Amount' for easy comparison\npremium_by_age_gender_desc = premium_by_age_gender.sort_values(by='Premium Amount', ascending=False)\n\ndisplay(\"Premium Amount by Age Group and Gender\", premium_by_age_gender_desc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:18.40984Z","iopub.execute_input":"2024-12-25T06:15:18.41012Z","iopub.status.idle":"2024-12-25T06:15:18.52947Z","shell.execute_reply.started":"2024-12-25T06:15:18.410096Z","shell.execute_reply":"2024-12-25T06:15:18.528689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by Gender and Age Group and calculate the mean Premium Amount\npremium_by_age_gender = df_train.groupby(['Gender', 'Age Group'])['Premium Amount'].mean().reset_index(name=\"Premium Amount\")\n\n# Sort the results by 'Premium Amount' for easy comparison\npremium_by_age_gender_desc = premium_by_age_gender.sort_values(by='Premium Amount', ascending=False)\n\n# Define a custom color palette for gender\ngender_palette = ['#B77A00', '#001F2D', '#29002d', '#b85712', '#34b1eb', '#eb3434', '#61eb34', '#776f7a', '#ff7d03']\n\n# Create the figure and axes\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(20, 10))\n\n# --- Bar Plot Enhancements ---\nsns.barplot(x='Premium Amount', y='Age Group', hue='Gender', data=premium_by_age_gender_desc, palette=gender_palette, ax=ax1)\n\n# Add values above the bars in ax1 with styling adjustments\nfor p in ax1.patches:\n    ax1.annotate(f'{p.get_width():.2f}', (p.get_x() + p.get_width() + 0.02, p.get_y() + p.get_height() / 2.),\n                 ha='left', va='center', fontsize=9, color='black', fontweight='bold')\n\n# Add title and labels for the barplot with custom font size and weight\nax1.set_title('Premium Amount by Age Group and Gender', fontsize=20, fontweight='bold', color='#2F4F4F')\nax1.set_xlabel('Mean Premium Amount', fontsize=14)\nax1.set_ylabel('Age Group', fontsize=14)\nax1.legend(title='Gender', loc='upper left', bbox_to_anchor=(1.05, 1), fontsize=12, title_fontsize=14)\n\n# Customize the grid and background for the bar plot\nax1.set_facecolor('#f5f5f5')\nax1.grid(True, axis='x', linestyle='--', alpha=0.7)\n\n# --- Pie Chart Enhancements ---\ngender_counts = df_train['Gender'].value_counts()\n\n# Add shadow and highlight the largest slice in the pie chart\nexplode = (0.1, 0) if len(gender_counts) > 1 else (0, 0)  # Exploding the first slice if more than one\nax2.pie(gender_counts, labels=gender_counts.index, autopct='%1.1f%%', startangle=90, colors=gender_palette, \n        wedgeprops={'edgecolor': 'black', 'linewidth': 2, 'linestyle': 'solid'}, explode=explode)\n\n# Customize the pie chart text to be white, bold, and larger\nfor text in ax2.texts:\n    text.set_fontsize(16)\n    text.set_color('white')\n    text.set_fontweight('bold')\n\n# Add a glowing effect to the pie chart title\nax2.set_title('Gender Distribution', fontsize=20, color='#FF4500', fontweight='bold')\n\n# --- Final Adjustments ---\nplt.tight_layout()\n\n# Display the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:26.21682Z","iopub.execute_input":"2024-12-25T06:15:26.217261Z","iopub.status.idle":"2024-12-25T06:15:27.21838Z","shell.execute_reply.started":"2024-12-25T06:15:26.217224Z","shell.execute_reply":"2024-12-25T06:15:27.217459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the data\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id')\nsample = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n# Feature Engineering for Date-related Features\ndf_train['Policy Start Date'] = pd.to_datetime(df_train['Policy Start Date'])\ndf_test['Policy Start Date'] = pd.to_datetime(df_test['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:30.688445Z","iopub.execute_input":"2024-12-25T06:15:30.688842Z","iopub.status.idle":"2024-12-25T06:15:37.078322Z","shell.execute_reply.started":"2024-12-25T06:15:30.68881Z","shell.execute_reply":"2024-12-25T06:15:37.077359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Year'] = df_train['Policy Start Date'].dt.year\ndf_test['Year'] = df_test['Policy Start Date'].dt.year\n\ndf_train['Month'] = df_train['Policy Start Date'].dt.month\ndf_test['Month'] = df_test['Policy Start Date'].dt.month\n\ndf_train['Day'] = df_train['Policy Start Date'].dt.day\ndf_test['Day'] = df_test['Policy Start Date'].dt.day\n\ndf_train['Month_name'] = df_train['Policy Start Date'].dt.month_name()\ndf_test['Month_name'] = df_test['Policy Start Date'].dt.month_name()\n\ndf_train['Day_of_week'] = df_train['Policy Start Date'].dt.day_name()\ndf_test['Day_of_week'] = df_test['Policy Start Date'].dt.day_name()\n\ndf_train['Week'] = df_train['Policy Start Date'].dt.isocalendar().week\ndf_test['Week'] = df_test['Policy Start Date'].dt.isocalendar().week\n\ndf_train['Year_sin'] = np.sin(2 * np.pi * df_train['Year'])\ndf_test['Year_sin'] = np.sin(2 * np.pi * df_test['Year'])\n\ndf_train['Year_cos'] = np.cos(2 * np.pi * df_train['Year'])\ndf_test['Year_cos'] = np.cos(2 * np.pi * df_test['Year'])\n\ndf_train['Month_sin'] = np.sin(2 * np.pi * df_train['Month'] / 12)\ndf_test['Month_sin'] = np.sin(2 * np.pi * df_test['Month'] / 12)\n\ndf_train['Month_cos'] = np.cos(2 * np.pi * df_train['Month'] / 12)\ndf_test['Month_cos'] = np.cos(2 * np.pi * df_test['Month'] / 12)\n\ndf_train['Day_sin'] = np.sin(2 * np.pi * df_train['Day'] / 31)\ndf_test['Day_sin'] = np.sin(2 * np.pi * df_test['Day'] / 31)\n\ndf_train['Day_cos'] = np.cos(2 * np.pi * df_train['Day'] / 31)\ndf_test['Day_cos'] = np.cos(2 * np.pi * df_test['Day'] / 31)\n\ndf_train['Group'] = (df_train['Year'] - 2020) * 48 + df_train['Month'] * 4 + df_train['Day'] // 7\ndf_test['Group'] = (df_test['Year'] - 2020) * 48 + df_test['Month'] * 4 + df_test['Day'] // 7\n\n# Drop the original 'Policy Start Date' column\ndf_train.drop('Policy Start Date', axis=1, inplace=True)\ndf_test.drop('Policy Start Date', axis=1, inplace=True)\n\n# Fill missing values in categorical columns\ncat_c = [col for col in df_train.columns if df_train[col].dtype == 'object']\n\nfor c in cat_c:\n    df_train[c] = df_train[c].fillna('missing').astype('category')\n    df_test[c] = df_test[c].fillna('missing').astype('category')\n\n# Convert 'Insurance Duration' to contract length categories\ndf_train['contract length'] = pd.cut(\n    df_train['Insurance Duration'].fillna(99), \n    bins=[-float('inf'), 1, 3, float('inf')], \n    labels=[0, 1, 2]\n).astype(int)\n\ndf_test['contract length'] = pd.cut(\n    df_test['Insurance Duration'].fillna(99), \n    bins=[-float('inf'), 1, 3, float('inf')], \n    labels=[0, 1, 2]\n).astype(int)\n\n# Log-transform the target variable 'Premium Amount'\ndf_train['Premium Amount'] = np.log(df_train['Premium Amount'])\n\n# Fill missing values in numerical columns\ndf_train['Age'] = df_train['Age'].fillna(df_train['Age'].median())\ndf_test['Age'] = df_test['Age'].fillna(df_train['Age'].median())\n\ndf_train['Annual Income'] = df_train['Annual Income'].fillna(df_train['Annual Income'].median())\ndf_test['Annual Income'] = df_test['Annual Income'].fillna(df_train['Annual Income'].median())\n\ndf_train['Number of Dependents'] = df_train['Number of Dependents'].fillna(df_train['Number of Dependents'].mode()[0])\ndf_test['Number of Dependents'] = df_test['Number of Dependents'].fillna(df_train['Number of Dependents'].mode()[0])\n\ndf_train['Health Score'] = df_train['Health Score'].fillna(df_train['Health Score'].median())\ndf_test['Health Score'] = df_test['Health Score'].fillna(df_train['Health Score'].median())\n\ndf_train['Previous Claims'] = df_train['Previous Claims'].fillna(df_train['Previous Claims'].mode()[0])\ndf_test['Previous Claims'] = df_test['Previous Claims'].fillna(df_train['Previous Claims'].mode()[0])\n\ndf_train['Vehicle Age'] = df_train['Vehicle Age'].fillna(df_train['Vehicle Age'].median())\ndf_test['Vehicle Age'] = df_test['Vehicle Age'].fillna(df_train['Vehicle Age'].median())\n\ndf_train['Credit Score'] = df_train['Credit Score'].fillna(df_train['Credit Score'].median())\ndf_test['Credit Score'] = df_test['Credit Score'].fillna(df_train['Credit Score'].median())\n\ndf_train['Insurance Duration'] = df_train['Insurance Duration'].fillna(df_train['Insurance Duration'].median())\ndf_test['Insurance Duration'] = df_test['Insurance Duration'].fillna(df_train['Insurance Duration'].median())\n\ndf_train['Customer Feedback'] = df_train['Customer Feedback'].fillna(df_train['Customer Feedback'].mode()[0])\ndf_test['Customer Feedback'] = df_test['Customer Feedback'].fillna(df_train['Customer Feedback'].mode()[0])\n\n# Check missing values again\nprint(\"Missing values present in Training Data:\", df_train.isnull().sum())\nprint(\"Missing values present in Testing Data:\", df_test.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:37.079711Z","iopub.execute_input":"2024-12-25T06:15:37.079991Z","iopub.status.idle":"2024-12-25T06:15:42.19827Z","shell.execute_reply.started":"2024-12-25T06:15:37.079967Z","shell.execute_reply":"2024-12-25T06:15:42.197333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters\nn_splits = 5\ny = df_train['Premium Amount']\nX = df_train.drop(['Premium Amount'], axis=1)\n\n# LightGBM Parameters for GPU\nlgb_params = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'boosting_type': 'gbdt',\n    'device_type': 'gpu',\n    'learning_rate': 0.01,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'random_state': 42\n}\n\n# CatBoost Parameters for GPU\ncb_params = {\n    'iterations': 1000,\n    'learning_rate': 0.01,\n    'loss_function': 'RMSE',\n    'random_seed': 42,\n    'early_stopping_rounds': 50,\n    'task_type': 'GPU',\n    'devices': '0'\n}\n\n# The rest of the code remains the same...\n\n\n\n# Initialize predictions\nlgb_predictions = np.zeros(len(df_test))\ncb_predictions = np.zeros(len(df_test))\n\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n# RMSLE function\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, np.maximum(y_pred, 0)))\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X)):\n    print(f\"Fold {fold + 1}\")\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    # LightGBM\n    lgb_model = lgb.LGBMRegressor(**lgb_params)\n    lgb_model.fit(X_train, y_train, eval_set=[(X_val, y_val)])\n    lgb_val_pred = lgb_model.predict(X_val)\n    print(f\"LightGBM Fold {fold + 1} RMSLE: {rmsle(y_val, lgb_val_pred):.4f}\")\n    lgb_predictions += lgb_model.predict(df_test) / n_splits\n\n    # CatBoost\n    cat_features = [X.columns.get_loc(col) for col in X.select_dtypes(include=['category', 'object']).columns]\n    cb_model = CatBoostRegressor(**cb_params)\n    cb_model.fit(X_train, y_train, eval_set=(X_val, y_val), cat_features=cat_features, verbose=100)\n    cb_val_pred = cb_model.predict(X_val)\n    print(f\"CatBoost Fold {fold + 1} RMSLE: {rmsle(y_val, cb_val_pred):.4f}\")\n    cb_predictions += cb_model.predict(df_test) / n_splits\n\n# Print final RMSLE for both models\nprint(f\"LightGBM Final RMSLE: {rmsle(y, lgb_model.predict(X)):.4f}\")\nprint(f\"CatBoost Final RMSLE: {rmsle(y, cb_model.predict(X)):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:15:51.406449Z","iopub.execute_input":"2024-12-25T06:15:51.406737Z","iopub.status.idle":"2024-12-25T06:28:48.995968Z","shell.execute_reply.started":"2024-12-25T06:15:51.406715Z","shell.execute_reply":"2024-12-25T06:28:48.994975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_predictions = (lgb_predictions + cb_predictions) / 2\nsample_submission=pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsample_submission['Premium Amount'] = np.exp(final_predictions)  # Revert log-transformation\nsample_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:28:48.997197Z","iopub.execute_input":"2024-12-25T06:28:48.997504Z","iopub.status.idle":"2024-12-25T06:28:50.453404Z","shell.execute_reply.started":"2024-12-25T06:28:48.997481Z","shell.execute_reply":"2024-12-25T06:28:50.452436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submisison=pd.read_csv(\"submission.csv\")\nsubmisison.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:28:50.454881Z","iopub.execute_input":"2024-12-25T06:28:50.455181Z","iopub.status.idle":"2024-12-25T06:28:50.646049Z","shell.execute_reply.started":"2024-12-25T06:28:50.455155Z","shell.execute_reply":"2024-12-25T06:28:50.645288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submisison.to_csv('modified_file.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T06:33:12.461825Z","iopub.execute_input":"2024-12-25T06:33:12.46222Z","iopub.status.idle":"2024-12-25T06:33:13.784057Z","shell.execute_reply.started":"2024-12-25T06:33:12.462189Z","shell.execute_reply":"2024-12-25T06:33:13.783083Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Observations\n\n### 1. Exercise Frequency and Related Metrics\n- **Highest Average Frequency Count**:\n  - **Weekly Exercise Frequency**: **0.255149**.\n  - Associated Mean Vehicle Age: **41.157696**.\n- **Lowest Exercise Frequency Count**:\n  - **Rarely Exercise Frequency**: **0.249517**.\n  - Associated Mean Vehicle Age: **41.148154**.\n- **Premium Counts by Exercise Frequency**:\n  - **Maximum Mean Premium Count**:\n    - **Daily Exercise Frequency**: **1,103.789093**.\n    - Mean Vehicle Age: **41.147298**.\n  - **Lowest Mean Premium Count**:\n    - **Weekly Exercise Frequency**: **1,101.234252**.\n    - Mean Vehicle Age: **41.157696**.\n- **Exercise Frequency Count Summary**:\n  - **Highest**: Weekly Exercise (**0.255**).\n  - **Lowest**: Monthly and Rarely Exercise (**0.250**).\n\n### 2. Premium Amounts by Vehicle Age\n- **Highest Premium Amount**:\n  - Vehicles aged **26–30**: **1,109.868955**.\n- **Lowest Premium Amount**:\n  - Vehicles aged **56–60**: **1,098.738754**.\n- **Maximum Premium Amount**:\n  - Vehicles aged **18–22**:\n    - Average Premium: **1,099.176374**.\n    - Count: **99,992**.\n    - Maximum Premium: **4,999.0**.\n- **Minimum Premium Amount**:\n  - Vehicles aged **46–50**:\n    - Average Premium: **1,101.450641**.\n    - Count: **126,349**.\n    - Maximum Premium: **4,992.0**.\n- **Vehicle Age Group with Maximum Count**:\n  - Age Range **26–30**:\n    - Count: **122,828**.\n    - Mean Premium: **1,109.868955**.\n\n### 3. Gender Proportions and Premium Analysis\n- **Gender Proportions**:\n  - **Females**: **50.2%**.\n  - **Males**: **49.8%**.\n- **Highest Premium by Gender and Age Range**:\n  - **Females (26–30)**: **1,110.17**.\n  - **Males (26–30)**: **1,109.57**.\n- **Lowest Premium by Gender and Age Range**:\n  - **Females (61–64)**: **1,096.26**.\n  - **Males (18–22)**: **1,093.74**.\n\n### 4. Total Premium Amounts by Exercise Frequency\n- **Highest Total Premium Amount**:\n  - **Exercise Frequency**: **Weekly**.\n  - Total Premium: **337,174,802.0**.\n- **Lowest Total Premium Amount**:\n  - **Exercise Frequency**: **Daily**.\n  - Total Premium: **325,144,557.0**.\n\n### 5. Machine Learning Models and Ensemble\n- Implemented **LGBM** and **CatBoost** models.\n- **Ensemble Method** yielded the **best results**.\n\n---\n\nThis organized structure improves readability and highlights critical insights effectively. Let me know if you'd like further refinements or additional context!\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}