{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport seaborn as sns\nimport matplotlib.pyplot as plt;\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:56:25.123199Z","iopub.execute_input":"2024-12-14T00:56:25.123588Z","iopub.status.idle":"2024-12-14T00:56:25.531203Z","shell.execute_reply.started":"2024-12-14T00:56:25.123557Z","shell.execute_reply":"2024-12-14T00:56:25.529588Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"** Insurance Premium Prediction Dataset **\n\n# Problem Statement:\nThe goal of this dataset is to facilitate the development and testing of regression models for predicting insurance premiums based on various customer characteristics and policy details. Insurance companies often rely on data-driven approaches to estimate premiums, taking into account factors such as age, income, health status, and claim history. This synthetic dataset simulates real-world scenarios to help practitioners practice feature engineering, data cleaning, and model training.\n\n# Dataset Overview\nThis dataset contains 2Lk+ and 20 features with a mix of categorical, numerical, and text data. It includes missing values, incorrect data types, and skewed distributions to mimic the complexities faced in real-world datasets. The target variable for prediction is the \"Premium Amount\".","metadata":{}},{"cell_type":"markdown","source":"# Dataset Understanding","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:28:35.094269Z","iopub.execute_input":"2024-12-14T00:28:35.094683Z","iopub.status.idle":"2024-12-14T00:28:46.235365Z","shell.execute_reply.started":"2024-12-14T00:28:35.094635Z","shell.execute_reply":"2024-12-14T00:28:46.234151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:12.911278Z","iopub.execute_input":"2024-12-14T00:54:12.911694Z","iopub.status.idle":"2024-12-14T00:54:12.938859Z","shell.execute_reply.started":"2024-12-14T00:54:12.91166Z","shell.execute_reply":"2024-12-14T00:54:12.936919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.tail(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:13.728282Z","iopub.execute_input":"2024-12-14T00:54:13.728728Z","iopub.status.idle":"2024-12-14T00:54:13.752856Z","shell.execute_reply.started":"2024-12-14T00:54:13.728691Z","shell.execute_reply":"2024-12-14T00:54:13.751811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"datsest info {df_train.info()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:14.23249Z","iopub.execute_input":"2024-12-14T00:54:14.232892Z","iopub.status.idle":"2024-12-14T00:54:14.899258Z","shell.execute_reply.started":"2024-12-14T00:54:14.232858Z","shell.execute_reply":"2024-12-14T00:54:14.898101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:14.901089Z","iopub.execute_input":"2024-12-14T00:54:14.901559Z","iopub.status.idle":"2024-12-14T00:54:15.626134Z","shell.execute_reply.started":"2024-12-14T00:54:14.901519Z","shell.execute_reply":"2024-12-14T00:54:15.624613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.describe(include = 'object').T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:15.62732Z","iopub.execute_input":"2024-12-14T00:54:15.627655Z","iopub.status.idle":"2024-12-14T00:54:17.868306Z","shell.execute_reply.started":"2024-12-14T00:54:15.627622Z","shell.execute_reply":"2024-12-14T00:54:17.867059Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Descriptive Overview of the Typical Client:\n\n* **Demographics:** Middle-aged individual (average age around 41 years), typically with two dependents\n\n* **Financial Profile:** Annual income averaging approximately $51,032, with a moderate credit score (around 592)\n  \n* **Vehicle Ownership:** Owns a vehicle averaging about 9.5 years old\n  \n* **Insurance History:** Insured for an average of 5 years, with a typical history of previous claims\n  \n* **Economic Characteristics:** Moderate income bracket with a slightly lower than average credit rating\n  \n* **Premium and Claims:** Average insurance premium of about $1,102, with variability in previous insurance claims* ","metadata":{}},{"cell_type":"code","source":"df_train.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:17.870416Z","iopub.execute_input":"2024-12-14T00:54:17.870764Z","iopub.status.idle":"2024-12-14T00:54:19.608054Z","shell.execute_reply.started":"2024-12-14T00:54:17.870732Z","shell.execute_reply":"2024-12-14T00:54:19.606881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:19.609137Z","iopub.execute_input":"2024-12-14T00:54:19.609419Z","iopub.status.idle":"2024-12-14T00:54:20.673128Z","shell.execute_reply.started":"2024-12-14T00:54:19.609392Z","shell.execute_reply":"2024-12-14T00:54:20.671321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"training dataset shape: {df_train.shape}\")\nprint(f\"testing dataset: {df_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:20.674532Z","iopub.execute_input":"2024-12-14T00:54:20.674902Z","iopub.status.idle":"2024-12-14T00:54:20.682879Z","shell.execute_reply.started":"2024-12-14T00:54:20.674868Z","shell.execute_reply":"2024-12-14T00:54:20.681332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:20.684672Z","iopub.execute_input":"2024-12-14T00:54:20.685179Z","iopub.status.idle":"2024-12-14T00:54:20.702664Z","shell.execute_reply.started":"2024-12-14T00:54:20.685128Z","shell.execute_reply":"2024-12-14T00:54:20.701396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:20.704005Z","iopub.execute_input":"2024-12-14T00:54:20.704442Z","iopub.status.idle":"2024-12-14T00:54:20.722163Z","shell.execute_reply.started":"2024-12-14T00:54:20.704401Z","shell.execute_reply":"2024-12-14T00:54:20.720873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:20.723585Z","iopub.execute_input":"2024-12-14T00:54:20.72408Z","iopub.status.idle":"2024-12-14T00:54:21.4137Z","shell.execute_reply.started":"2024-12-14T00:54:20.724038Z","shell.execute_reply":"2024-12-14T00:54:21.412706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Wrangling","metadata":{}},{"cell_type":"code","source":"#check missing values\ndf_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:21.416441Z","iopub.execute_input":"2024-12-14T00:54:21.416922Z","iopub.status.idle":"2024-12-14T00:54:22.051625Z","shell.execute_reply.started":"2024-12-14T00:54:21.416886Z","shell.execute_reply":"2024-12-14T00:54:22.050688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:22.052927Z","iopub.execute_input":"2024-12-14T00:54:22.053273Z","iopub.status.idle":"2024-12-14T00:54:22.479077Z","shell.execute_reply.started":"2024-12-14T00:54:22.053241Z","shell.execute_reply":"2024-12-14T00:54:22.477895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Premium Amount'].isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:22.480594Z","iopub.execute_input":"2024-12-14T00:54:22.481078Z","iopub.status.idle":"2024-12-14T00:54:22.491042Z","shell.execute_reply.started":"2024-12-14T00:54:22.481028Z","shell.execute_reply":"2024-12-14T00:54:22.489841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#find the percentage of missing values\nmissing_data = df_train.isnull().mean() * 100\nfor col, pct in missing_data.items():\n    print(f\"{col} - {pct:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:22.492283Z","iopub.execute_input":"2024-12-14T00:54:22.492754Z","iopub.status.idle":"2024-12-14T00:54:23.132872Z","shell.execute_reply.started":"2024-12-14T00:54:22.492721Z","shell.execute_reply":"2024-12-14T00:54:23.131373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = df_train.drop(['id'], axis =1)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:23.134605Z","iopub.execute_input":"2024-12-14T00:54:23.135086Z","iopub.status.idle":"2024-12-14T00:54:23.324845Z","shell.execute_reply.started":"2024-12-14T00:54:23.135038Z","shell.execute_reply":"2024-12-14T00:54:23.323737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test = df_test.drop(['id'], axis=1)\ntest = df_test.copy()\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:23.326354Z","iopub.execute_input":"2024-12-14T00:54:23.326771Z","iopub.status.idle":"2024-12-14T00:54:23.439267Z","shell.execute_reply.started":"2024-12-14T00:54:23.326725Z","shell.execute_reply":"2024-12-14T00:54:23.438034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:23.440648Z","iopub.execute_input":"2024-12-14T00:54:23.44113Z","iopub.status.idle":"2024-12-14T00:54:23.448348Z","shell.execute_reply.started":"2024-12-14T00:54:23.441084Z","shell.execute_reply":"2024-12-14T00:54:23.447121Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Handle MIssing Values","metadata":{}},{"cell_type":"code","source":"#Numeric Columns\nnumeric_records = train.select_dtypes(include=np.number)\nnumeric_records","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:23.449724Z","iopub.execute_input":"2024-12-14T00:54:23.45009Z","iopub.status.idle":"2024-12-14T00:54:23.522222Z","shell.execute_reply.started":"2024-12-14T00:54:23.450059Z","shell.execute_reply":"2024-12-14T00:54:23.521171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_numerical_features(df):\n    \"\"\"\n    Preprocess the numerical features by filling in missings.\n\n    Args:\n        df: pandas dataframe\n\n    Returns: pandas dataframe with filled missings\n    \n    \"\"\"\n    df = df.copy()\n    \n    df['Number of Dependents'].fillna(df['Number of Dependents'].median(), inplace=True)\n    df['Age'].fillna(df['Age'].median(), inplace=True)\n    df['Vehicle Age'].fillna(df['Vehicle Age'].median(), inplace=True)\n    df['Previous Claims'].fillna(df['Previous Claims'].median(), inplace=True)\n    df['Annual Income'].fillna(df['Annual Income'].median(), inplace=True)\n    df['Credit Score'].fillna(df['Credit Score'].median(), inplace=True)\n    df['Insurance Duration'].fillna(df['Insurance Duration'].median(), inplace=True)\n    df['Health Score'].fillna(df['Health Score'].median(), inplace=True)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:23.523403Z","iopub.execute_input":"2024-12-14T00:54:23.523728Z","iopub.status.idle":"2024-12-14T00:54:23.53106Z","shell.execute_reply.started":"2024-12-14T00:54:23.523695Z","shell.execute_reply":"2024-12-14T00:54:23.529863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute numerical columns\ntrain = preprocess_numerical_features(train)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:24.945746Z","iopub.execute_input":"2024-12-14T00:54:24.946157Z","iopub.status.idle":"2024-12-14T00:54:25.406779Z","shell.execute_reply.started":"2024-12-14T00:54:24.946121Z","shell.execute_reply":"2024-12-14T00:54:25.405557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = preprocess_numerical_features(test)\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:25.919667Z","iopub.execute_input":"2024-12-14T00:54:25.920076Z","iopub.status.idle":"2024-12-14T00:54:26.23089Z","shell.execute_reply.started":"2024-12-14T00:54:25.920043Z","shell.execute_reply":"2024-12-14T00:54:26.229635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_policy_start_date_features(df):\n\n    df = df.copy()\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'], errors='coerce')\n    df['Policy Start Year'] = df['Policy Start Date'].dt.year\n    df['Policy Start Month'] = df['Policy Start Date'].dt.month\n    df['Policy Start Day'] = df['Policy Start Date'].dt.day\n    df['Policy Start DayOfWeek'] = df['Policy Start Date'].dt.dayofweek\n    df['Policy Tenure'] = (pd.Timestamp.now() - df['Policy Start Date']).dt.days / 365.25\n\n    df = df.drop(columns=['Policy Start Date'])\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:27.924532Z","iopub.execute_input":"2024-12-14T00:54:27.925Z","iopub.status.idle":"2024-12-14T00:54:27.931429Z","shell.execute_reply.started":"2024-12-14T00:54:27.924964Z","shell.execute_reply":"2024-12-14T00:54:27.930335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = get_policy_start_date_features(test)\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:30.093871Z","iopub.execute_input":"2024-12-14T00:54:30.094295Z","iopub.status.idle":"2024-12-14T00:54:30.954429Z","shell.execute_reply.started":"2024-12-14T00:54:30.094262Z","shell.execute_reply":"2024-12-14T00:54:30.953352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = get_policy_start_date_features(train)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:30.956081Z","iopub.execute_input":"2024-12-14T00:54:30.956403Z","iopub.status.idle":"2024-12-14T00:54:32.205778Z","shell.execute_reply.started":"2024-12-14T00:54:30.956373Z","shell.execute_reply":"2024-12-14T00:54:32.204602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Categorical Columns\ncategorical_records = train.select_dtypes(include='object')\ncategorical_records","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:33.334834Z","iopub.execute_input":"2024-12-14T00:54:33.335364Z","iopub.status.idle":"2024-12-14T00:54:33.86566Z","shell.execute_reply.started":"2024-12-14T00:54:33.335295Z","shell.execute_reply":"2024-12-14T00:54:33.864501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_records.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:35.329324Z","iopub.execute_input":"2024-12-14T00:54:35.329714Z","iopub.status.idle":"2024-12-14T00:54:35.895156Z","shell.execute_reply.started":"2024-12-14T00:54:35.329679Z","shell.execute_reply":"2024-12-14T00:54:35.893852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Customer Feedback'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:37.053169Z","iopub.execute_input":"2024-12-14T00:54:37.053634Z","iopub.status.idle":"2024-12-14T00:54:37.113164Z","shell.execute_reply.started":"2024-12-14T00:54:37.053595Z","shell.execute_reply":"2024-12-14T00:54:37.111973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_records.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:54:38.025158Z","iopub.execute_input":"2024-12-14T00:54:38.025599Z","iopub.status.idle":"2024-12-14T00:54:38.033268Z","shell.execute_reply.started":"2024-12-14T00:54:38.025551Z","shell.execute_reply":"2024-12-14T00:54:38.032037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"plt.style.use('ggplot')\nsns.kdeplot(data=train, x='Premium Amount', color='steelblue', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:04:36.672194Z","iopub.execute_input":"2024-12-14T00:04:36.672869Z","iopub.status.idle":"2024-12-14T00:04:42.715166Z","shell.execute_reply.started":"2024-12-14T00:04:36.672816Z","shell.execute_reply":"2024-12-14T00:04:42.71396Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From above figure the distribution is right-skewed. Next, we explore potential relationships between the input features and `Premium Amount`","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.scatterplot(data=train, x='Age', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Gender', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:04:42.717304Z","iopub.execute_input":"2024-12-14T00:04:42.71785Z","iopub.status.idle":"2024-12-14T00:04:48.57117Z","shell.execute_reply.started":"2024-12-14T00:04:42.717811Z","shell.execute_reply":"2024-12-14T00:04:48.570056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Number of Dependents', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Marital Status', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:04:48.572609Z","iopub.execute_input":"2024-12-14T00:04:48.572978Z","iopub.status.idle":"2024-12-14T00:04:50.065717Z","shell.execute_reply.started":"2024-12-14T00:04:48.572944Z","shell.execute_reply":"2024-12-14T00:04:50.064557Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes.\n\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Education Level', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Occupation', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:04:50.067123Z","iopub.execute_input":"2024-12-14T00:04:50.067447Z","iopub.status.idle":"2024-12-14T00:04:51.865856Z","shell.execute_reply.started":"2024-12-14T00:04:50.067415Z","shell.execute_reply":"2024-12-14T00:04:51.864706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above charts, there is no interesting relationship that can be exploited for modeling purposes.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.scatterplot(data=train, x='Health Score', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Location', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:16:37.479322Z","iopub.execute_input":"2024-12-13T23:16:37.479671Z","iopub.status.idle":"2024-12-13T23:16:43.475214Z","shell.execute_reply.started":"2024-12-13T23:16:37.479637Z","shell.execute_reply":"2024-12-13T23:16:43.473894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Previous Claims', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Policy Type', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:16:43.476893Z","iopub.execute_input":"2024-12-13T23:16:43.477334Z","iopub.status.idle":"2024-12-13T23:16:45.050394Z","shell.execute_reply.started":"2024-12-13T23:16:43.477288Z","shell.execute_reply":"2024-12-13T23:16:45.049201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(18, 7))\n\nplt_1 = sns.boxplot(data=train, x='Vehicle Age', y='Premium Amount', ax=ax[0])\nplt_2 = sns.scatterplot(data=train, x='Credit Score', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:16:45.051725Z","iopub.execute_input":"2024-12-13T23:16:45.052054Z","iopub.status.idle":"2024-12-13T23:16:50.903031Z","shell.execute_reply.started":"2024-12-13T23:16:45.052024Z","shell.execute_reply":"2024-12-13T23:16:50.901822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(18, 7))\n\nplt_1 = sns.scatterplot(data=train, x='Annual Income', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Property Type', y='Premium Amount', ax=ax[1]);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:16:50.904971Z","iopub.execute_input":"2024-12-13T23:16:50.905329Z","iopub.status.idle":"2024-12-13T23:16:56.830555Z","shell.execute_reply.started":"2024-12-13T23:16:50.905296Z","shell.execute_reply":"2024-12-13T23:16:56.82943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[numeric_records.columns].hist(bins=30, figsize=(12, 15))\nplt.suptitle('Histograms of Numerical Columns')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:16:56.832018Z","iopub.execute_input":"2024-12-13T23:16:56.832356Z","iopub.status.idle":"2024-12-13T23:16:59.302312Z","shell.execute_reply.started":"2024-12-13T23:16:56.832322Z","shell.execute_reply":"2024-12-13T23:16:59.300892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in categorical_records.columns:\n    plt.figure(figsize=(4, 3))\n    sns.countplot(x=col, data=train)\n    plt.title(f'Count Plot of {col}')\n    plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:16:59.306033Z","iopub.execute_input":"2024-12-13T23:16:59.306554Z","iopub.status.idle":"2024-12-13T23:17:07.752296Z","shell.execute_reply.started":"2024-12-13T23:16:59.306503Z","shell.execute_reply":"2024-12-13T23:17:07.750977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = train[numeric_records.columns].corr()\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, cmap = 'Blues')\nplt.title('Correlation Heatmap of Numerical Columns')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T23:17:07.753845Z","iopub.execute_input":"2024-12-13T23:17:07.754312Z","iopub.status.idle":"2024-12-13T23:17:08.473917Z","shell.execute_reply.started":"2024-12-13T23:17:07.754263Z","shell.execute_reply":"2024-12-13T23:17:08.472712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PreProcessing","metadata":{}},{"cell_type":"markdown","source":"## Handle Missing Values","metadata":{}},{"cell_type":"code","source":"categorical_records.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:32.743341Z","iopub.execute_input":"2024-12-14T00:55:32.743741Z","iopub.status.idle":"2024-12-14T00:55:32.751863Z","shell.execute_reply.started":"2024-12-14T00:55:32.743705Z","shell.execute_reply":"2024-12-14T00:55:32.75069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_records.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:33.234765Z","iopub.execute_input":"2024-12-14T00:55:33.235184Z","iopub.status.idle":"2024-12-14T00:55:33.798778Z","shell.execute_reply.started":"2024-12-14T00:55:33.235146Z","shell.execute_reply":"2024-12-14T00:55:33.797482Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## preprocess_categorical_features","metadata":{}},{"cell_type":"code","source":"def preprocess_categorical_features(df):\n    \"\"\"\n    Preprocess the categorical features by filling in missings.\n\n    Args:\n        df: pandas dataframe\n\n    Returns: pandas dataframe with filled missings\n    \n    \"\"\"\n\n    df = df.copy()\n    df['Marital Status'] = df['Marital Status'].fillna(df['Marital Status'].mode()[0])\n    df['Occupation'] = df['Occupation'].fillna('Unkown')\n    df['Customer Feedback'] = df['Customer Feedback'].fillna('Unknown')\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:35.102756Z","iopub.execute_input":"2024-12-14T00:55:35.10372Z","iopub.status.idle":"2024-12-14T00:55:35.10973Z","shell.execute_reply.started":"2024-12-14T00:55:35.103678Z","shell.execute_reply":"2024-12-14T00:55:35.108339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_processed = preprocess_categorical_features(train)\ntrain_processed.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:36.106776Z","iopub.execute_input":"2024-12-14T00:55:36.10779Z","iopub.status.idle":"2024-12-14T00:55:37.139498Z","shell.execute_reply.started":"2024-12-14T00:55:36.107744Z","shell.execute_reply":"2024-12-14T00:55:37.13736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_processed = preprocess_categorical_features(test)\ntest_processed.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:37.141831Z","iopub.execute_input":"2024-12-14T00:55:37.142386Z","iopub.status.idle":"2024-12-14T00:55:37.787763Z","shell.execute_reply.started":"2024-12-14T00:55:37.142334Z","shell.execute_reply":"2024-12-14T00:55:37.786695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.select_dtypes(include=['object']).isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:37.918317Z","iopub.execute_input":"2024-12-14T00:55:37.918728Z","iopub.status.idle":"2024-12-14T00:55:38.984276Z","shell.execute_reply.started":"2024-12-14T00:55:37.918688Z","shell.execute_reply":"2024-12-14T00:55:38.982979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.select_dtypes(include=['object']).isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:38.986044Z","iopub.execute_input":"2024-12-14T00:55:38.98639Z","iopub.status.idle":"2024-12-14T00:55:39.66558Z","shell.execute_reply.started":"2024-12-14T00:55:38.98636Z","shell.execute_reply":"2024-12-14T00:55:39.664547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Label Encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\n\ndef df_after_label_encoding(df):\n    df = df.copy()\n    df['Education Level'] = label_encoder.fit_transform(df['Education Level'])\n    df['Occupation'] = label_encoder.fit_transform(df['Occupation'])\n    df['Policy Type'] = label_encoder.fit_transform(df['Policy Type'])\n    # One-Hot Encoding for nominal data\n    df = pd.get_dummies(df, columns=['Gender', 'Location', 'Marital Status', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'Occupation', 'Policy Type'], drop_first=True)\n    # df['Policy Type'].map(df.groupby('Policy Type')['Premium Amount'].mean())\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:44.259524Z","iopub.execute_input":"2024-12-14T00:55:44.259935Z","iopub.status.idle":"2024-12-14T00:55:44.266853Z","shell.execute_reply.started":"2024-12-14T00:55:44.2599Z","shell.execute_reply":"2024-12-14T00:55:44.265573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_encoded = df_after_label_encoding(train_processed)\ntrain_encoded.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:47.699237Z","iopub.execute_input":"2024-12-14T00:55:47.699648Z","iopub.status.idle":"2024-12-14T00:55:49.689917Z","shell.execute_reply.started":"2024-12-14T00:55:47.699611Z","shell.execute_reply":"2024-12-14T00:55:49.688791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train['Education Level'][:4].unique)\ncategorical_records['Education Level'][:4].unique","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:55.516144Z","iopub.execute_input":"2024-12-14T00:55:55.516516Z","iopub.status.idle":"2024-12-14T00:55:55.526186Z","shell.execute_reply.started":"2024-12-14T00:55:55.516485Z","shell.execute_reply":"2024-12-14T00:55:55.524795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_encoded = df_after_label_encoding(test_processed)\ntest_encoded.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:55:55.939163Z","iopub.execute_input":"2024-12-14T00:55:55.939555Z","iopub.status.idle":"2024-12-14T00:55:57.101981Z","shell.execute_reply.started":"2024-12-14T00:55:55.939516Z","shell.execute_reply":"2024-12-14T00:55:57.100819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.lineplot(x='Policy Start Day', y='Premium Amount', data=train)\nplt.xlabel('Policy Start Day')\nplt.ylabel('Target Amount')\nplt.title('Daily Patterns of Premium Amount')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:56:37.927153Z","iopub.execute_input":"2024-12-14T00:56:37.927664Z","iopub.status.idle":"2024-12-14T00:56:47.099415Z","shell.execute_reply.started":"2024-12-14T00:56:37.92763Z","shell.execute_reply":"2024-12-14T00:56:47.098175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Policy Start Month', y='Premium Amount', data=train)\nplt.xlabel('Policy Start Month')\nplt.ylabel('Target Amount')\nplt.title('Seasonality of Target Amount by Month')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:56:47.101231Z","iopub.execute_input":"2024-12-14T00:56:47.101576Z","iopub.status.idle":"2024-12-14T00:56:47.65067Z","shell.execute_reply.started":"2024-12-14T00:56:47.101543Z","shell.execute_reply":"2024-12-14T00:56:47.64949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.lineplot(x='Policy Start Year', y='Premium Amount', data=train)\nplt.xlabel('Policy Start Year')\nplt.ylabel('Premium Amount')\nplt.title('Trend of Premium Amount Over Years')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:56:47.652164Z","iopub.execute_input":"2024-12-14T00:56:47.652596Z","iopub.status.idle":"2024-12-14T00:57:00.569199Z","shell.execute_reply.started":"2024-12-14T00:56:47.652547Z","shell.execute_reply":"2024-12-14T00:57:00.568044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handle Outliers\nOutliers are data points significantly different from the rest of the dataset. Identifying and handling outliers is crucial because they can skew model performance and impact the interpretation of results","metadata":{}},{"cell_type":"code","source":"\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef draw_box_plot(df, column, title, xlabel):\n    sns.boxplot(x=df[column])\n    \n    plt.title(title)\n    plt.xlabel(xlabel)\n    \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:17.315279Z","iopub.execute_input":"2024-12-14T00:05:17.315985Z","iopub.status.idle":"2024-12-14T00:05:17.323432Z","shell.execute_reply.started":"2024-12-14T00:05:17.315932Z","shell.execute_reply":"2024-12-14T00:05:17.321718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def draw_box_plots(df, columns, titles, xlabels):\n    # Create a figure with 1 row and len(columns) columns\n    fig, axes = plt.subplots(1, len(columns), figsize=(6, 5))\n    \n    # If there's only one column, convert axes to a list to make it iterable\n    if len(columns) == 1:\n        axes = [axes]\n    \n    # Create box plots for each column\n    for i, (column, title, xlabel) in enumerate(zip(columns, titles, xlabels)):\n        sns.boxplot(x=df[column], ax=axes[i])\n        axes[i].set_title(title)\n        axes[i].set_xlabel(xlabel)\n    \n    # Adjust layout to prevent overlap\n    plt.tight_layout()\n    \n    # Show the plot\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:17.326314Z","iopub.execute_input":"2024-12-14T00:05:17.327006Z","iopub.status.idle":"2024-12-14T00:05:17.341772Z","shell.execute_reply.started":"2024-12-14T00:05:17.32694Z","shell.execute_reply":"2024-12-14T00:05:17.340495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def draw_multiple_box_plots(df, columns):\n    # Calculate number of rows needed (2 columns per row)\n    num_rows = (len(columns) + 1) // 2\n    \n    # Create figure with appropriate size\n    fig, axes = plt.subplots(num_rows, 2, figsize=(20, 5 * num_rows))\n    \n    # Flatten axes for easy iteration\n    axes = axes.flatten()\n    \n    # Create box plots\n    for i, column in enumerate(columns):\n        sns.boxplot(x=df[column], ax=axes[i])\n        axes[i].set_title(f'Boxplot of {column}')\n        axes[i].set_xlabel(column)\n    \n    # Remove extra subplots\n    for j in range(i+1, len(axes)):\n        fig.delaxes(axes[j])\n    \n    # Adjust layout\n    plt.tight_layout()\n    \n    # Show the plot\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:17.343153Z","iopub.execute_input":"2024-12-14T00:05:17.343594Z","iopub.status.idle":"2024-12-14T00:05:17.361751Z","shell.execute_reply.started":"2024-12-14T00:05:17.343548Z","shell.execute_reply":"2024-12-14T00:05:17.360495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"draw_multiple_box_plots(train, [\n    'Annual Income', \n    'Number of Dependents', \n    'Health Score', \n    'Previous Claims', \n    'Vehicle Age', \n    'Credit Score', \n    'Insurance Duration'\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:17.363317Z","iopub.execute_input":"2024-12-14T00:05:17.363641Z","iopub.status.idle":"2024-12-14T00:05:19.075859Z","shell.execute_reply.started":"2024-12-14T00:05:17.36361Z","shell.execute_reply":"2024-12-14T00:05:19.074731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outliers_iqr(data, column):\n    Q1 = data[column].quantile(0.25)\n    Q3 = data[column].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    outliers = data[(data[column] < lower_bound) | (data[column] > upper_bound)]\n    return outliers, lower_bound, upper_bound\n# outliers, lower_bound, upper_bound=detect_outliers_iqr(train, 'Annual Income')\n# outliers.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.077391Z","iopub.execute_input":"2024-12-14T00:05:19.077759Z","iopub.status.idle":"2024-12-14T00:05:19.083995Z","shell.execute_reply.started":"2024-12-14T00:05:19.077725Z","shell.execute_reply":"2024-12-14T00:05:19.08283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_outliers(df, columns):\n    outlier_summary = {}\n    \n    for column in columns:\n        # Calculate IQR\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n        \n        # Calculate bounds\n        lower_bound = Q1 - (1.5 * IQR)\n        upper_bound = Q3 + (1.5 * IQR)\n        \n        # Identify outliers\n        outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n        \n        outlier_summary[column] = {\n            'total_rows': len(df),\n            'outliers_count': len(outliers),\n            'outliers_percentage': (len(outliers) / len(df)) * 100,\n            'lower_bound': lower_bound,\n            'upper_bound': upper_bound\n        }\n    \n    return outlier_summary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.085301Z","iopub.execute_input":"2024-12-14T00:05:19.085772Z","iopub.status.idle":"2024-12-14T00:05:19.095256Z","shell.execute_reply.started":"2024-12-14T00:05:19.085726Z","shell.execute_reply":"2024-12-14T00:05:19.094265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyze_outliers(train, numeric_records.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.098729Z","iopub.execute_input":"2024-12-14T00:05:19.099312Z","iopub.status.idle":"2024-12-14T00:05:19.603866Z","shell.execute_reply.started":"2024-12-14T00:05:19.09926Z","shell.execute_reply":"2024-12-14T00:05:19.6026Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are three approaches to dealing with outliers:\n* Trimming/Remove the outliers\n* Quantile Based Flooring and Capping\n  - In this technique, the outlier is capped at a certain value above the 90th percentile value or floored at a factor below the 10th percentile value.\n* Mean/Median Imputation\n  - As the mean value is highly influenced by the outlier treatment, it is advised to replace the outliers with the median value.","metadata":{}},{"cell_type":"code","source":"def outlier_data_imputation(data, column, method=\"mean\"):\n    outliers, lower_bound, upper_bound = detect_outliers_iqr(data, column)\n    if not len(outliers):\n        print({\n        \"method\": method,\n        \"impute_value\": None,\n        \"outliers_count\":  len(outliers),\n        \"column\": column\n        })\n        return data\n    elif method == \"mean\":\n        impute_value = data[column].mean()\n    elif method == \"median\":\n        impute_value = data[column].median()\n    else:\n        raise ValueError(\"Method must be 'mean' or 'median'\")\n    print({\n        \"method\": method,\n        \"impute_value\": impute_value,\n        \"outliers_count\":  len(outliers),\n        \"column\": column\n        })\n    #imputation\n    data.loc[(data[column] < lower_bound) | (data[column] > upper_bound), column] = impute_value\n\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.605235Z","iopub.execute_input":"2024-12-14T00:05:19.605546Z","iopub.status.idle":"2024-12-14T00:05:19.61322Z","shell.execute_reply.started":"2024-12-14T00:05:19.605515Z","shell.execute_reply":"2024-12-14T00:05:19.612066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_imputation_impact(original_df, processed_df, columns, imputation_method):\n    import matplotlib.pyplot as plt\n    import seaborn as sns\n    \n    # Create a figure with subplots\n    fig, axes = plt.subplots(len(columns), 2, figsize=(15, 4*len(columns)))\n    \n    for i, column in enumerate(columns):\n        # Original distribution\n        sns.boxplot(x=original_df[column], ax=axes[i, 0])\n        axes[i, 0].set_title(f'Original {column}')\n        \n        # Processed distribution\n        sns.boxplot(x=processed_df[column], ax=axes[i, 1])\n        axes[i, 1].set_title(f'{imputation_method.capitalize()} Imputed {column}')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.614571Z","iopub.execute_input":"2024-12-14T00:05:19.614985Z","iopub.status.idle":"2024-12-14T00:05:19.627089Z","shell.execute_reply.started":"2024-12-14T00:05:19.614926Z","shell.execute_reply":"2024-12-14T00:05:19.625972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def comprehensive_outlier_handling(df, columns, imputation_method='mean'):\n    df_processed = df.copy()\n    for column in columns:\n        df_processed = outlier_data_imputation(df_processed, column, imputation_method)\n    visualize_imputation_impact(df, df_processed, columns, imputation_method)\n    return df_processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.628579Z","iopub.execute_input":"2024-12-14T00:05:19.628957Z","iopub.status.idle":"2024-12-14T00:05:19.645883Z","shell.execute_reply.started":"2024-12-14T00:05:19.628926Z","shell.execute_reply":"2024-12-14T00:05:19.644737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_processed = comprehensive_outlier_handling(train, numeric_records.columns, \"mean\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:19.647086Z","iopub.execute_input":"2024-12-14T00:05:19.647444Z","iopub.status.idle":"2024-12-14T00:05:24.75767Z","shell.execute_reply.started":"2024-12-14T00:05:19.647409Z","shell.execute_reply":"2024-12-14T00:05:24.75658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_processed_median = comprehensive_outlier_handling(train, numeric_records.columns, \"median\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:05:24.75896Z","iopub.execute_input":"2024-12-14T00:05:24.759283Z","iopub.status.idle":"2024-12-14T00:05:29.676833Z","shell.execute_reply.started":"2024-12-14T00:05:24.75925Z","shell.execute_reply":"2024-12-14T00:05:29.675465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Normalize/Scale Numerical Features","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\ntrain_encoded[numeric_records.columns] = scaler.fit_transform(train_encoded[numeric_records.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:57:12.576131Z","iopub.execute_input":"2024-12-14T00:57:12.576531Z","iopub.status.idle":"2024-12-14T00:57:12.721261Z","shell.execute_reply.started":"2024-12-14T00:57:12.576488Z","shell.execute_reply":"2024-12-14T00:57:12.720095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_numeric_cols = test_encoded.select_dtypes(include=np.number).columns.tolist()\ntest_categoric_cols = test_encoded.select_dtypes(include=['object']).columns.tolist()\ntest_encoded[test_numeric_cols] = scaler.fit_transform(test_encoded[test_numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T00:57:17.547576Z","iopub.execute_input":"2024-12-14T00:57:17.548009Z","iopub.status.idle":"2024-12-14T00:57:17.772276Z","shell.execute_reply.started":"2024-12-14T00:57:17.547942Z","shell.execute_reply":"2024-12-14T00:57:17.771192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"X = train_encoded.drop(columns=['Premium Amount', 'Customer Feedback'])\ny = train_encoded['Premium Amount']\ny_log = np.log1p(train_encoded['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:01:42.053318Z","iopub.execute_input":"2024-12-14T01:01:42.053734Z","iopub.status.idle":"2024-12-14T01:01:42.168739Z","shell.execute_reply.started":"2024-12-14T01:01:42.053698Z","shell.execute_reply":"2024-12-14T01:01:42.167735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_columns = X.select_dtypes(include=np.number).columns.tolist()\ncategorical_columns = X.select_dtypes(include=['object']).columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:01:43.371313Z","iopub.execute_input":"2024-12-14T01:01:43.371702Z","iopub.status.idle":"2024-12-14T01:01:43.489376Z","shell.execute_reply.started":"2024-12-14T01:01:43.37167Z","shell.execute_reply":"2024-12-14T01:01:43.488339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y_log, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:01:51.758101Z","iopub.execute_input":"2024-12-14T01:01:51.758471Z","iopub.status.idle":"2024-12-14T01:01:52.191056Z","shell.execute_reply.started":"2024-12-14T01:01:51.75844Z","shell.execute_reply":"2024-12-14T01:01:52.190034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\n# Initialize the model\nmodel = LinearRegression()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:01:52.646507Z","iopub.execute_input":"2024-12-14T01:01:52.646971Z","iopub.status.idle":"2024-12-14T01:01:52.652152Z","shell.execute_reply.started":"2024-12-14T01:01:52.64692Z","shell.execute_reply":"2024-12-14T01:01:52.651002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:01:53.974141Z","iopub.execute_input":"2024-12-14T01:01:53.974539Z","iopub.status.idle":"2024-12-14T01:01:56.259694Z","shell.execute_reply.started":"2024-12-14T01:01:53.974504Z","shell.execute_reply":"2024-12-14T01:01:56.258552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the test set\ny_pred = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:02:09.704148Z","iopub.execute_input":"2024-12-14T01:02:09.704532Z","iopub.status.idle":"2024-12-14T01:02:09.75202Z","shell.execute_reply.started":"2024-12-14T01:02:09.704481Z","shell.execute_reply":"2024-12-14T01:02:09.750646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_original = np.expm1(y_pred)\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\n\nmae = mean_absolute_error(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"MAE: {mae:.2f}\")\nprint(f\"MSE: {mse:.2f}\")\nprint(f\"R² Score: {r2:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:02:35.665376Z","iopub.execute_input":"2024-12-14T01:02:35.66578Z","iopub.status.idle":"2024-12-14T01:02:35.682303Z","shell.execute_reply.started":"2024-12-14T01:02:35.665747Z","shell.execute_reply":"2024-12-14T01:02:35.680984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For Linear Regression\ncoefficients = pd.DataFrame({\n    'Feature': X.columns,\n    'Coefficient': model.coef_\n})\nprint(coefficients.sort_values(by='Coefficient', ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:02:53.025034Z","iopub.execute_input":"2024-12-14T01:02:53.025431Z","iopub.status.idle":"2024-12-14T01:02:53.037306Z","shell.execute_reply.started":"2024-12-14T01:02:53.025396Z","shell.execute_reply":"2024-12-14T01:02:53.035653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(X_train.corr(), annot=True, cmap='coolwarm')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:10:36.953275Z","iopub.execute_input":"2024-12-14T01:10:36.953799Z","iopub.status.idle":"2024-12-14T01:10:41.832264Z","shell.execute_reply.started":"2024-12-14T01:10:36.953752Z","shell.execute_reply":"2024-12-14T01:10:41.830942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nrf_model = RandomForestRegressor(random_state=42)\nrf_model.fit(X_train, y_train)\nrf_pred = rf_model.predict(X_test)\n\n# Evaluate the Random Forest model\nmae_rf = mean_absolute_error(y_test, rf_pred)\nprint(f\"Random Forest MAE: {mae_rf:.2f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import PolynomialFeatures\n\npoly = PolynomialFeatures(degree=2)\nX_train_poly = poly.fit_transform(X_train)\nX_test_poly = poly.transform(X_test)\n\npoly_model = LinearRegression()\npoly_model.fit(X_train_poly, y_train)\ny_pred_poly = poly_model.predict(X_test_poly)\n\nprint(f\"R² Score (Polynomial Regression): {r2_score(y_test, y_pred_poly):.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:11:35.43801Z","iopub.execute_input":"2024-12-14T01:11:35.438863Z","iopub.status.idle":"2024-12-14T01:12:40.909097Z","shell.execute_reply.started":"2024-12-14T01:11:35.4388Z","shell.execute_reply":"2024-12-14T01:12:40.907031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:13:36.73222Z","iopub.execute_input":"2024-12-14T01:13:36.732807Z","iopub.status.idle":"2024-12-14T01:13:50.040014Z","shell.execute_reply.started":"2024-12-14T01:13:36.732763Z","shell.execute_reply":"2024-12-14T01:13:50.038097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n# Initialize the XGBRegressor\nxgb_model = XGBRegressor(\n    n_estimators=100,  # Number of trees\n    learning_rate=0.1,  # Learning rate\n    max_depth=5,  # Maximum depth of a tree\n    random_state=42  # For reproducibility\n)\n\n# Fit the model on training data\nxgb_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:13:59.170734Z","iopub.execute_input":"2024-12-14T01:13:59.171235Z","iopub.status.idle":"2024-12-14T01:14:07.512828Z","shell.execute_reply.started":"2024-12-14T01:13:59.171195Z","shell.execute_reply":"2024-12-14T01:14:07.51156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the test set\ny_pred = xgb_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:14:24.743601Z","iopub.execute_input":"2024-12-14T01:14:24.744805Z","iopub.status.idle":"2024-12-14T01:14:25.451735Z","shell.execute_reply.started":"2024-12-14T01:14:24.744736Z","shell.execute_reply":"2024-12-14T01:14:25.450485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_original = np.expm1(y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:14:35.54787Z","iopub.execute_input":"2024-12-14T01:14:35.548316Z","iopub.status.idle":"2024-12-14T01:14:35.555593Z","shell.execute_reply.started":"2024-12-14T01:14:35.54828Z","shell.execute_reply":"2024-12-14T01:14:35.554538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\n\n# Calculate metrics\nmae = mean_absolute_error(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nrmse = np.sqrt(mse)\nr2 = r2_score(y_test, y_pred)\n\n# Print results\nprint(f\"MAE: {mae:.2f}\")\nprint(f\"MSE: {mse:.2f}\")\nprint(f\"RMSE: {rmse:.2f}\")\nprint(f\"R² Score: {r2:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:14:51.218546Z","iopub.execute_input":"2024-12-14T01:14:51.219002Z","iopub.status.idle":"2024-12-14T01:14:51.235234Z","shell.execute_reply.started":"2024-12-14T01:14:51.218964Z","shell.execute_reply":"2024-12-14T01:14:51.23373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\n# Define parameter grid\nparam_grid = {\n    'n_estimators': [50, 100, 200],\n    'learning_rate': [0.01, 0.1, 0.2],\n    'max_depth': [3, 5, 7],\n    'subsample': [0.8, 1.0]\n}\n\n# Initialize Grid Search\ngrid_search = GridSearchCV(\n    estimator=XGBRegressor(random_state=42),\n    param_grid=param_grid,\n    scoring='neg_mean_absolute_error',\n    cv=3,\n    verbose=1\n)\n\n# Fit Grid Search on training data\ngrid_search.fit(X_train, y_train)\n\n# Best parameters\nprint(\"Best Parameters:\", grid_search.best_params_)\n\n# Use the best estimator\nbest_xgb_model = grid_search.best_estimator_\n\n# Evaluate the best model\ny_pred_best = best_xgb_model.predict(X_test)\nprint(f\"Best RMSE: {np.sqrt(mean_squared_error(y_test, y_pred_best)):.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:15:44.577113Z","iopub.execute_input":"2024-12-14T01:15:44.577802Z","iopub.status.idle":"2024-12-14T01:34:46.211718Z","shell.execute_reply.started":"2024-12-14T01:15:44.577743Z","shell.execute_reply":"2024-12-14T01:34:46.210851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import MinMaxScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:34:46.213199Z","iopub.execute_input":"2024-12-14T01:34:46.2135Z","iopub.status.idle":"2024-12-14T01:34:46.235599Z","shell.execute_reply.started":"2024-12-14T01:34:46.213472Z","shell.execute_reply":"2024-12-14T01:34:46.234435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Numerical preprocessing\nnumerical_preprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', MinMaxScaler())\n])\nnumerical_preprocessor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:34:46.237019Z","iopub.execute_input":"2024-12-14T01:34:46.237367Z","iopub.status.idle":"2024-12-14T01:34:46.250249Z","shell.execute_reply.started":"2024-12-14T01:34:46.237335Z","shell.execute_reply":"2024-12-14T01:34:46.249119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categorical preprocessing\ncategorical_preprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('one_hot_encoder', OneHotEncoder(drop='first', sparse_output=False))\n])\ncategorical_preprocessor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:34:46.253225Z","iopub.execute_input":"2024-12-14T01:34:46.253744Z","iopub.status.idle":"2024-12-14T01:34:46.271639Z","shell.execute_reply.started":"2024-12-14T01:34:46.253692Z","shell.execute_reply":"2024-12-14T01:34:46.270535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(transformers=[\n    ('num', numerical_preprocessor, numeric_columns),  # Numerical processing\n    ('cat', categorical_preprocessor, categorical_columns)  # Categorical processing\n])\npreprocessor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:34:46.273065Z","iopub.execute_input":"2024-12-14T01:34:46.27339Z","iopub.status.idle":"2024-12-14T01:34:46.311133Z","shell.execute_reply.started":"2024-12-14T01:34:46.273359Z","shell.execute_reply":"2024-12-14T01:34:46.309779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Full pipeline with model\nmodel = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('regressor', RandomForestRegressor(random_state=42))\n])\nmodel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:34:46.31294Z","iopub.execute_input":"2024-12-14T01:34:46.313421Z","iopub.status.idle":"2024-12-14T01:34:46.387528Z","shell.execute_reply.started":"2024-12-14T01:34:46.313374Z","shell.execute_reply":"2024-12-14T01:34:46.385532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nmodel.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T01:34:46.388926Z","iopub.execute_input":"2024-12-14T01:34:46.389331Z","iopub.status.idle":"2024-12-14T02:08:18.111198Z","shell.execute_reply.started":"2024-12-14T01:34:46.389298Z","shell.execute_reply":"2024-12-14T02:08:18.10951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}