{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA) – Insurance Dataset\n\n## 🔍 Overview\nThis notebook presents a **structured and competition-oriented Exploratory Data Analysis (EDA)** of an insurance dataset.  \nThe analysis focuses on understanding data distributions, detecting anomalies, handling missing values, and uncovering meaningful patterns that can directly support **feature engineering and model performance**.\n\n## 🎯 Objectives\n- Analyze numerical and categorical feature distributions  \n- Identify and assess missing values and potential data quality issues  \n- Detect outliers and evaluate their impact on analysis and modeling  \n- Explore correlations and key relationships between variables  \n- Generate actionable insights for downstream machine learning tasks  \n\n## 🛠 Tools & Techniques\n- **Python:** Pandas, NumPy  \n- **Visualization:** Matplotlib, Seaborn  \n- **Statistical Analysis:** Descriptive statistics, IQR-based outlier detection  \n\nThis notebook follows **best practices used in Kaggle competitions** and aims to build a solid analytical foundation for predictive modeling.\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport random\nimport numpy as np\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_absolute_error, r2_score\nfrom sklearn.preprocessing import LabelEncoder , OneHotEncoder , MinMaxScaler, StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:21.477972Z","iopub.execute_input":"2026-02-06T16:42:21.478658Z","iopub.status.idle":"2026-02-06T16:42:21.485247Z","shell.execute_reply.started":"2026-02-06T16:42:21.478624Z","shell.execute_reply":"2026-02-06T16:42:21.483872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:21.487396Z","iopub.execute_input":"2026-02-06T16:42:21.487923Z","iopub.status.idle":"2026-02-06T16:42:21.508832Z","shell.execute_reply.started":"2026-02-06T16:42:21.487887Z","shell.execute_reply":"2026-02-06T16:42:21.507416Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Step 1 - Reading the Data**\n","metadata":{}},{"cell_type":"markdown","source":"# tasks\n### Calculate the percentaage of missing values ?¶\n### Handle data types\n### Plotting the data ( EDA ) to explore potential handling techniques\n### Detect the outlier\n### Explore How to visulaize the Date and time data in column ( Policy Start Date )","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv') # Loading the insurance dataset into a DataFrame\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:21.510208Z","iopub.execute_input":"2026-02-06T16:42:21.510753Z","iopub.status.idle":"2026-02-06T16:42:26.732431Z","shell.execute_reply.started":"2026-02-06T16:42:21.510727Z","shell.execute_reply":"2026-02-06T16:42:26.731577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:26.733582Z","iopub.execute_input":"2026-02-06T16:42:26.733806Z","iopub.status.idle":"2026-02-06T16:42:27.465623Z","shell.execute_reply.started":"2026-02-06T16:42:26.733789Z","shell.execute_reply":"2026-02-06T16:42:27.464347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info() # Checking for null values and data types","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:27.467807Z","iopub.execute_input":"2026-02-06T16:42:27.468108Z","iopub.status.idle":"2026-02-06T16:42:28.186952Z","shell.execute_reply.started":"2026-02-06T16:42:27.468086Z","shell.execute_reply":"2026-02-06T16:42:28.18584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **step 2 Detecting the proplems !** \n ## 1-  Dropping columns as it is not useful for analysis","metadata":{}},{"cell_type":"code","source":"df.drop(columns=['id'], inplace=True)  # Dropping 'id' column as it is not useful for analysis\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:28.188041Z","iopub.execute_input":"2026-02-06T16:42:28.188388Z","iopub.status.idle":"2026-02-06T16:42:29.110445Z","shell.execute_reply.started":"2026-02-06T16:42:28.188357Z","shell.execute_reply":"2026-02-06T16:42:29.109286Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2- converting date and time columns to 'datetime64'","metadata":{}},{"cell_type":"code","source":"df['Policy Start Date'] = df['Policy Start Date'].astype('datetime64[ns]')\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:29.111679Z","iopub.execute_input":"2026-02-06T16:42:29.112645Z","iopub.status.idle":"2026-02-06T16:42:30.387527Z","shell.execute_reply.started":"2026-02-06T16:42:29.112617Z","shell.execute_reply":"2026-02-06T16:42:30.386498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3- Converting categorical columns to 'category' data type\n  ### and Assignment categorical columns to a variable 'cat'\n  ### and Assignment numerical columns to a variable 'num'\n   ","metadata":{}},{"cell_type":"code","source":"cat = df.select_dtypes(include='object').columns # Selecting categorical columns\nnum = df.select_dtypes(include=['int64', 'float64']).columns # Selecting numerical columns\nprint(\"Categorical :\", cat)\nprint(\"Numerical:\", num) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:30.388487Z","iopub.execute_input":"2026-02-06T16:42:30.388801Z","iopub.status.idle":"2026-02-06T16:42:30.930788Z","shell.execute_reply.started":"2026-02-06T16:42:30.388767Z","shell.execute_reply":"2026-02-06T16:42:30.929925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in cat: # Converting categorical columns to 'category' data type\n    df[i] = df[i].astype('category')\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:30.931633Z","iopub.execute_input":"2026-02-06T16:42:30.932011Z","iopub.status.idle":"2026-02-06T16:42:31.882108Z","shell.execute_reply.started":"2026-02-06T16:42:30.931964Z","shell.execute_reply":"2026-02-06T16:42:31.880772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:31.883383Z","iopub.execute_input":"2026-02-06T16:42:31.883742Z","iopub.status.idle":"2026-02-06T16:42:31.891095Z","shell.execute_reply.started":"2026-02-06T16:42:31.883713Z","shell.execute_reply":"2026-02-06T16:42:31.890078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Checking unique values in some columns . if it likes a categorical column or not\nprint(df['Health Score'].value_counts()) \n\"\"\" Checking unique values in 'Health Score' column .\ndf['Health Score'] = df['Health Score'].astype('category') \nit has a wide range of values \"\"\"\nprint(df['Credit Score'].value_counts()) \n\"\"\" Checking unique values in 'Credit Score' column .\n#df['Credit Score'] = df['Credit Score'].astype('category') it has a wide range of values \"\"\"\n\nprint(df['Number of Dependents'].value_counts()) # Checking unique values in 'Number of Dependents' column .\n#df['Number of Dependents'] = df['Number of Dependents'].astype('category') # we can convert it to categorical as it has limited unique values\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:31.892397Z","iopub.execute_input":"2026-02-06T16:42:31.892842Z","iopub.status.idle":"2026-02-06T16:42:32.172271Z","shell.execute_reply.started":"2026-02-06T16:42:31.892742Z","shell.execute_reply":"2026-02-06T16:42:32.170824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4- Checking duplicate < non duplicate values>","metadata":{}},{"cell_type":"code","source":"df.duplicated().sum() # Checking for duplicate rows in the dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:32.173871Z","iopub.execute_input":"2026-02-06T16:42:32.174233Z","iopub.status.idle":"2026-02-06T16:42:33.078381Z","shell.execute_reply.started":"2026-02-06T16:42:32.174212Z","shell.execute_reply":"2026-02-06T16:42:33.077217Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5- null values < we have meny null values>","metadata":{}},{"cell_type":"code","source":"df.isnull().sum() # Checking for null values in each column","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:33.079773Z","iopub.execute_input":"2026-02-06T16:42:33.080609Z","iopub.status.idle":"2026-02-06T16:42:33.165747Z","shell.execute_reply.started":"2026-02-06T16:42:33.080581Z","shell.execute_reply":"2026-02-06T16:42:33.16469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## >>Calculate the percentaage of missing values ?","metadata":{}},{"cell_type":"code","source":"count_agenull = df['Age'].isnull().sum()   # Calculate the percentaage of missing values for Age column\npercentaage =  count_agenull / df.shape[0] * 100\npercentaage ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:33.169827Z","iopub.execute_input":"2026-02-06T16:42:33.170236Z","iopub.status.idle":"2026-02-06T16:42:33.183669Z","shell.execute_reply.started":"2026-02-06T16:42:33.170207Z","shell.execute_reply":"2026-02-06T16:42:33.182485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:33.184959Z","iopub.execute_input":"2026-02-06T16:42:33.185327Z","iopub.status.idle":"2026-02-06T16:42:33.196562Z","shell.execute_reply.started":"2026-02-06T16:42:33.18529Z","shell.execute_reply":"2026-02-06T16:42:33.19531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# colect missing values py catgorical columns and numerical columns\ncat_missing = []\nnum_missing=[]\nfor i in cat:\n    if df[i].isnull().sum() > 0 :\n        cat_missing.append(i) \nfor i in num:\n    if df[i].isnull().sum() > 0 :\n        num_missing.append(i) \n\nprint(cat_missing)\nprint(num_missing)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:33.197664Z","iopub.execute_input":"2026-02-06T16:42:33.198033Z","iopub.status.idle":"2026-02-06T16:42:33.295174Z","shell.execute_reply.started":"2026-02-06T16:42:33.197976Z","shell.execute_reply":"2026-02-06T16:42:33.294122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# defind function Calculate the percentaage of missing values in col\n\ndef  Percen_DfNull(col):\n    count_agenull = df[col].isnull().sum()   \n    percentaage =  count_agenull / df.shape[0] * 100\n    return percentaage\nprint(\" percentaage null for  catgorical columns ********************************\")\nfor i in cat_missing:\n    print(i,\"= \",Percen_DfNull(i).round(4))\nprint(\" \\npercentaage null for  numerical columns ********************************\")\nfor i in num_missing:\n    print(i,\"= \",Percen_DfNull(i).round(4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:33.296134Z","iopub.execute_input":"2026-02-06T16:42:33.296539Z","iopub.status.idle":"2026-02-06T16:42:33.364119Z","shell.execute_reply.started":"2026-02-06T16:42:33.296491Z","shell.execute_reply":"2026-02-06T16:42:33.363134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### >>show Par plot for catgorical columns befor handling missing values add Code","metadata":{}},{"cell_type":"code","source":"for i in cat:\n    plt.figure(figsize=(8,5))\n    df[i].value_counts().plot(kind='bar', color='lightgreen')\n    plt.title(i)\n    plt.xlabel('catgory')\n    plt.ylabel('count')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:33.365229Z","iopub.execute_input":"2026-02-06T16:42:33.365554Z","iopub.status.idle":"2026-02-06T16:42:35.022932Z","shell.execute_reply.started":"2026-02-06T16:42:33.365527Z","shell.execute_reply":"2026-02-06T16:42:35.021624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8,5))\ndf['Number of Dependents'].value_counts().plot(kind='bar', color='lightgreen')\nplt.title('Number of Dependents')\nplt.xlabel('catgory')\nplt.ylabel('count')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:35.024422Z","iopub.execute_input":"2026-02-06T16:42:35.025463Z","iopub.status.idle":"2026-02-06T16:42:35.213928Z","shell.execute_reply.started":"2026-02-06T16:42:35.025414Z","shell.execute_reply":"2026-02-06T16:42:35.212378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8,5))\ndf['Previous Claims'].value_counts().plot(kind='bar', color='lightgreen')\nplt.title('Previous Claims')\nplt.xlabel('catgory')\nplt.ylabel('count')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:35.215191Z","iopub.execute_input":"2026-02-06T16:42:35.215556Z","iopub.status.idle":"2026-02-06T16:42:35.441771Z","shell.execute_reply.started":"2026-02-06T16:42:35.215528Z","shell.execute_reply":"2026-02-06T16:42:35.440698Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### >>show histgram for numarical columns befor handling missing values","metadata":{}},{"cell_type":"code","source":"for i in num_missing:\n    plt.figure(figsize=(8,5))             \n    plt.hist(df[i], bins=30, color='skyblue', edgecolor='black') \n    plt.title(i)  \n    plt.xlabel('values')                       \n    plt.ylabel('counter')                       \n    plt.grid(axis='y', alpha=0.75)            \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:35.442835Z","iopub.execute_input":"2026-02-06T16:42:35.443082Z","iopub.status.idle":"2026-02-06T16:42:37.323945Z","shell.execute_reply.started":"2026-02-06T16:42:35.443063Z","shell.execute_reply":"2026-02-06T16:42:37.322851Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling missing Values for catgorical columns","metadata":{}},{"cell_type":"code","source":"# catgorical Columns\ndf['Marital Status'] = df['Marital Status'].fillna(df['Marital Status'].mode()[0])  #fill missing values py mode()[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.325041Z","iopub.execute_input":"2026-02-06T16:42:37.325506Z","iopub.status.idle":"2026-02-06T16:42:37.343789Z","shell.execute_reply.started":"2026-02-06T16:42:37.325477Z","shell.execute_reply":"2026-02-06T16:42:37.342475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# numerical Columns\ndf['Number of Dependents'] = df['Number of Dependents'].fillna(-1.0) #fill missing values py -1.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.344881Z","iopub.execute_input":"2026-02-06T16:42:37.345299Z","iopub.status.idle":"2026-02-06T16:42:37.365087Z","shell.execute_reply.started":"2026-02-06T16:42:37.345258Z","shell.execute_reply":"2026-02-06T16:42:37.363904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Number of Dependents'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.366413Z","iopub.execute_input":"2026-02-06T16:42:37.367156Z","iopub.status.idle":"2026-02-06T16:42:37.391292Z","shell.execute_reply.started":"2026-02-06T16:42:37.367131Z","shell.execute_reply":"2026-02-06T16:42:37.390411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# convert to categorical type\ndf['Number of Dependents'] = df['Number of Dependents'].astype('category') \ndf['Number of Dependents'].info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.392855Z","iopub.execute_input":"2026-02-06T16:42:37.393288Z","iopub.status.idle":"2026-02-06T16:42:37.447383Z","shell.execute_reply.started":"2026-02-06T16:42:37.393263Z","shell.execute_reply":"2026-02-06T16:42:37.446473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Occupation'] = df['Occupation'].cat.add_categories([\"Unknown\"]) # add new category","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.44843Z","iopub.execute_input":"2026-02-06T16:42:37.448753Z","iopub.status.idle":"2026-02-06T16:42:37.454872Z","shell.execute_reply.started":"2026-02-06T16:42:37.448726Z","shell.execute_reply":"2026-02-06T16:42:37.453864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Occupation'] = df['Occupation'].fillna('Unknown')  # fill missing values py new category","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.455881Z","iopub.execute_input":"2026-02-06T16:42:37.456375Z","iopub.status.idle":"2026-02-06T16:42:37.483407Z","shell.execute_reply.started":"2026-02-06T16:42:37.456307Z","shell.execute_reply":"2026-02-06T16:42:37.482394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Occupation'].isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.484274Z","iopub.execute_input":"2026-02-06T16:42:37.484611Z","iopub.status.idle":"2026-02-06T16:42:37.494153Z","shell.execute_reply.started":"2026-02-06T16:42:37.484579Z","shell.execute_reply":"2026-02-06T16:42:37.493163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(Percen_DfNull('Customer Feedback'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.495267Z","iopub.execute_input":"2026-02-06T16:42:37.49567Z","iopub.status.idle":"2026-02-06T16:42:37.511804Z","shell.execute_reply.started":"2026-02-06T16:42:37.495648Z","shell.execute_reply":"2026-02-06T16:42:37.510886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Customer Feedback'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.513035Z","iopub.execute_input":"2026-02-06T16:42:37.513636Z","iopub.status.idle":"2026-02-06T16:42:37.538629Z","shell.execute_reply.started":"2026-02-06T16:42:37.513609Z","shell.execute_reply":"2026-02-06T16:42:37.537804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fill nan values py MODE \ndf['Customer Feedback'] = df['Customer Feedback'].fillna(df['Customer Feedback'].mode()[0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.539614Z","iopub.execute_input":"2026-02-06T16:42:37.53993Z","iopub.status.idle":"2026-02-06T16:42:37.559155Z","shell.execute_reply.started":"2026-02-06T16:42:37.539902Z","shell.execute_reply":"2026-02-06T16:42:37.558138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Customer Feedback'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.560393Z","iopub.execute_input":"2026-02-06T16:42:37.560764Z","iopub.status.idle":"2026-02-06T16:42:37.571429Z","shell.execute_reply.started":"2026-02-06T16:42:37.560732Z","shell.execute_reply":"2026-02-06T16:42:37.570304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove 'Number of Dependents' and 'Previous Claims' from numerical columns list\nnum = num.drop(['Number of Dependents','Previous Claims'])\nnum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.572413Z","iopub.execute_input":"2026-02-06T16:42:37.572659Z","iopub.status.idle":"2026-02-06T16:42:37.58802Z","shell.execute_reply.started":"2026-02-06T16:42:37.572639Z","shell.execute_reply":"2026-02-06T16:42:37.586813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Previous Claims'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.589423Z","iopub.execute_input":"2026-02-06T16:42:37.589684Z","iopub.status.idle":"2026-02-06T16:42:37.631987Z","shell.execute_reply.started":"2026-02-06T16:42:37.589663Z","shell.execute_reply":"2026-02-06T16:42:37.630779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Previous Claims column is like catgorical column\n# first convert to object colem\n# than fill nan values \n\n#df['Previous Claims'] = df['Previous Claims'].astype('object')\ndf['Previous Claims'] = df['Previous Claims'].fillna(-1.0)\ndf['Previous Claims'] =df['Previous Claims'].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.633093Z","iopub.execute_input":"2026-02-06T16:42:37.633454Z","iopub.status.idle":"2026-02-06T16:42:37.677504Z","shell.execute_reply.started":"2026-02-06T16:42:37.633426Z","shell.execute_reply":"2026-02-06T16:42:37.676262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf['Previous Claims'].isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.678612Z","iopub.execute_input":"2026-02-06T16:42:37.678945Z","iopub.status.idle":"2026-02-06T16:42:37.688475Z","shell.execute_reply.started":"2026-02-06T16:42:37.678915Z","shell.execute_reply":"2026-02-06T16:42:37.686977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# append 'Number of Dependents' and 'Previous Claims' to catgorical columns\ncat=cat.append(pd.Index(['Number of Dependents','Previous Claims']))\ncat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.689626Z","iopub.execute_input":"2026-02-06T16:42:37.689864Z","iopub.status.idle":"2026-02-06T16:42:37.705449Z","shell.execute_reply.started":"2026-02-06T16:42:37.689844Z","shell.execute_reply":"2026-02-06T16:42:37.704237Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling missing Values for NUMERICAL columns","metadata":{}},{"cell_type":"code","source":"for i in num:\n    if i in num_missing:\n        print(f\"{i} null percent =\",Percen_DfNull(i))\n    print(df[i].value_counts())\n    print(\"***************************\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:37.706593Z","iopub.execute_input":"2026-02-06T16:42:37.706915Z","iopub.status.idle":"2026-02-06T16:42:38.0629Z","shell.execute_reply.started":"2026-02-06T16:42:37.706886Z","shell.execute_reply":"2026-02-06T16:42:38.062032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df['Age'] = df['Age'].fillna(df['Age'].mean())  # numerical\n#df['Annual Income'] = df['Annual Income'].fillna(df['Annual Income'].mean())  # numerical","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:38.063867Z","iopub.execute_input":"2026-02-06T16:42:38.064206Z","iopub.status.idle":"2026-02-06T16:42:38.069208Z","shell.execute_reply.started":"2026-02-06T16:42:38.064183Z","shell.execute_reply":"2026-02-06T16:42:38.067989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col = df['Health Score']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:38.070349Z","iopub.execute_input":"2026-02-06T16:42:38.070964Z","iopub.status.idle":"2026-02-06T16:42:38.091576Z","shell.execute_reply.started":"2026-02-06T16:42:38.070934Z","shell.execute_reply":"2026-02-06T16:42:38.09021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m = round(col.mean(),2)\nstd = round(col.std(),2)\nprint(std,m)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:38.098522Z","iopub.execute_input":"2026-02-06T16:42:38.100661Z","iopub.status.idle":"2026-02-06T16:42:38.138564Z","shell.execute_reply.started":"2026-02-06T16:42:38.100606Z","shell.execute_reply":"2026-02-06T16:42:38.137534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fill nan values py random values betwen m-std and m+std \nfor i in num:\n    if df[i].isnull().sum() > 0:\n        m = df[i].mean()\n        std = df[i].std()\n        missing_idx = df[df[i].isna()].index\n        rand_values = np.random.uniform(m - std, m + std, size=len(missing_idx))\n        df.loc[missing_idx, i] = rand_values\n        print(f\"{i}: Filled {len(missing_idx)} missing values\")\n    else:\n        print(f\"{i}: No missing values\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:38.139795Z","iopub.execute_input":"2026-02-06T16:42:38.141821Z","iopub.status.idle":"2026-02-06T16:42:38.379056Z","shell.execute_reply.started":"2026-02-06T16:42:38.141788Z","shell.execute_reply":"2026-02-06T16:42:38.378181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Step3 Visulization - Exploratory Data Analysis EDA¶**\n","metadata":{}},{"cell_type":"markdown","source":"## show histgram for numarical columns **after** handling missing values","metadata":{}},{"cell_type":"code","source":"for i in num:\n    plt.figure(figsize=(8,5))             \n    plt.hist(df[i], bins=30, color='skyblue', edgecolor='black') \n    plt.title(i)  \n    plt.xlabel('values')                       \n    plt.ylabel('count')                       \n    plt.grid(axis='y', alpha=0.75)            \n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:38.380069Z","iopub.execute_input":"2026-02-06T16:42:38.380299Z","iopub.status.idle":"2026-02-06T16:42:40.014652Z","shell.execute_reply.started":"2026-02-06T16:42:38.380281Z","shell.execute_reply":"2026-02-06T16:42:40.013789Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## show Par plot for catgorical columns  **after** handling missing values add Code","metadata":{}},{"cell_type":"code","source":"for i in cat:\n    plt.figure(figsize=(8,5))\n    df[i].value_counts().plot(kind='bar', color='lightgreen')\n    plt.title(i)\n    plt.xlabel('catgory')\n    plt.ylabel('count')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:40.015555Z","iopub.execute_input":"2026-02-06T16:42:40.015893Z","iopub.status.idle":"2026-02-06T16:42:41.959487Z","shell.execute_reply.started":"2026-02-06T16:42:40.015872Z","shell.execute_reply":"2026-02-06T16:42:41.958394Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# step3-4 we chick all reletionships between columns in a table","metadata":{}},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:41.960589Z","iopub.execute_input":"2026-02-06T16:42:41.960973Z","iopub.status.idle":"2026-02-06T16:42:41.968354Z","shell.execute_reply.started":"2026-02-06T16:42:41.960949Z","shell.execute_reply":"2026-02-06T16:42:41.96725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:41.96955Z","iopub.execute_input":"2026-02-06T16:42:41.969948Z","iopub.status.idle":"2026-02-06T16:42:41.988995Z","shell.execute_reply.started":"2026-02-06T16:42:41.969925Z","shell.execute_reply":"2026-02-06T16:42:41.987901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"fig = px.histogram(\n                    df,\n                    x= 'Premium Amount',\n                    y= 'Annual Income',\n                    color=\"Marital Status\",\n                    nbins=20,\n                    opacity=0.5,\n                    title=\"Premium Distribution and Annual Income by Marital Status\" )\n\nfig.show()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:41.99043Z","iopub.execute_input":"2026-02-06T16:42:41.991401Z","iopub.status.idle":"2026-02-06T16:42:42.008406Z","shell.execute_reply.started":"2026-02-06T16:42:41.991364Z","shell.execute_reply":"2026-02-06T16:42:42.00739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\" The proportions of the categories in the column 'Policy Type' are approximately equal, \n              therefore there is no direct or strong effect on the target column.\"\"\"\n\"\"\"fig = px.histogram(\n                    df,\n                    x='Premium Amount',\n                    color='Policy Type',\n                   \n                    nbins=20,\n                    opacity=0.5,\n                    title=\"Premium Distribution by Policy Type\" )\n\nfig.show()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.009482Z","iopub.execute_input":"2026-02-06T16:42:42.009797Z","iopub.status.idle":"2026-02-06T16:42:42.030538Z","shell.execute_reply.started":"2026-02-06T16:42:42.009769Z","shell.execute_reply":"2026-02-06T16:42:42.029269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\" The proportions of the categories in the column 'Gender' are approximately equal, \n              therefore there is no direct or strong effect on the target column.\"\"\"\n\"\"\"fig = px.histogram(\n    df,\n    x='Premium Amount',\n   \n    color=\"Gender\",\n    nbins=20,\n    opacity=0.5,\n    title=\"Premium Distribution by Gende\" )\n\nfig.show()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.031797Z","iopub.execute_input":"2026-02-06T16:42:42.032221Z","iopub.status.idle":"2026-02-06T16:42:42.05278Z","shell.execute_reply.started":"2026-02-06T16:42:42.032167Z","shell.execute_reply":"2026-02-06T16:42:42.051386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"fig = px.histogram(\n    df,\n    x='Premium Amount',\n    color=\"Insurance Duration\",\n    nbins=20,\n    opacity=0.5,\n    title=\"Premium Distribution by Insurance Duration\" )\n\nfig.show()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.053716Z","iopub.execute_input":"2026-02-06T16:42:42.054133Z","iopub.status.idle":"2026-02-06T16:42:42.069081Z","shell.execute_reply.started":"2026-02-06T16:42:42.054109Z","shell.execute_reply":"2026-02-06T16:42:42.068058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ins = df['Insurance Duration'].sort_values()\nins","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.070043Z","iopub.execute_input":"2026-02-06T16:42:42.070404Z","iopub.status.idle":"2026-02-06T16:42:42.205244Z","shell.execute_reply.started":"2026-02-06T16:42:42.070376Z","shell.execute_reply":"2026-02-06T16:42:42.204492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = px.histogram(\n    df,\n    x='Premium Amount',\n    color=ins,\n    nbins=20,\n    opacity=0.5,\n    title=\"Premium Distribution by Insurance Duration\" )\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.206156Z","iopub.execute_input":"2026-02-06T16:42:42.20649Z","iopub.status.idle":"2026-02-06T16:42:42.597824Z","shell.execute_reply.started":"2026-02-06T16:42:42.206455Z","shell.execute_reply":"2026-02-06T16:42:42.595951Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### There is no direct influence between 'Insurance Duration' and 'Premium Amount'","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.599144Z","iopub.execute_input":"2026-02-06T16:42:42.599654Z","iopub.status.idle":"2026-02-06T16:42:42.676216Z","shell.execute_reply.started":"2026-02-06T16:42:42.599608Z","shell.execute_reply":"2026-02-06T16:42:42.675129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.groupby('Previous Claims')['Policy Type'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.677446Z","iopub.execute_input":"2026-02-06T16:42:42.678461Z","iopub.status.idle":"2026-02-06T16:42:42.724915Z","shell.execute_reply.started":"2026-02-06T16:42:42.678422Z","shell.execute_reply":"2026-02-06T16:42:42.723783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"month = df['Policy Start Date'].dt.month\nplt.figure(figsize=(8,5))\ndf.groupby(month)['Premium Amount'].max().plot(marker='o', linestyle='-', color='teal')\nplt.title('Premium Amount py month')\nplt.xlabel('month')\nplt.ylabel('Premium Amount')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.726228Z","iopub.execute_input":"2026-02-06T16:42:42.727011Z","iopub.status.idle":"2026-02-06T16:42:42.996666Z","shell.execute_reply.started":"2026-02-06T16:42:42.726973Z","shell.execute_reply":"2026-02-06T16:42:42.995909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"year = df['Policy Start Date'].dt.year\nplt.figure(figsize=(8,5))\ndf.groupby(year)['Premium Amount'].max().plot(marker='o', linestyle='-', color='teal')\nplt.title('Premium Amount py year')\nplt.xlabel('year')\nplt.ylabel('Premium Amount')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:42.997574Z","iopub.execute_input":"2026-02-06T16:42:42.997861Z","iopub.status.idle":"2026-02-06T16:42:43.258886Z","shell.execute_reply.started":"2026-02-06T16:42:42.997841Z","shell.execute_reply":"2026-02-06T16:42:43.257785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.groupby(year)['Policy Type'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.260056Z","iopub.execute_input":"2026-02-06T16:42:43.260356Z","iopub.status.idle":"2026-02-06T16:42:43.316077Z","shell.execute_reply.started":"2026-02-06T16:42:43.260294Z","shell.execute_reply":"2026-02-06T16:42:43.314963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.317339Z","iopub.execute_input":"2026-02-06T16:42:43.31765Z","iopub.status.idle":"2026-02-06T16:42:43.324804Z","shell.execute_reply.started":"2026-02-06T16:42:43.317622Z","shell.execute_reply":"2026-02-06T16:42:43.323847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"for i in cat:\n     fig = px.pie(df,names= i , title=f'{i}')\n     fig.show()\n   \"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.325736Z","iopub.execute_input":"2026-02-06T16:42:43.32605Z","iopub.status.idle":"2026-02-06T16:42:43.344107Z","shell.execute_reply.started":"2026-02-06T16:42:43.326028Z","shell.execute_reply":"2026-02-06T16:42:43.343006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### All category columns have similar percentages for each category.","metadata":{}},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.345124Z","iopub.execute_input":"2026-02-06T16:42:43.3455Z","iopub.status.idle":"2026-02-06T16:42:43.362148Z","shell.execute_reply.started":"2026-02-06T16:42:43.345468Z","shell.execute_reply":"2026-02-06T16:42:43.360841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NumCorr=[]\nfor i in num:\n    NumCorr.append(df[i].corr(df['Premium Amount']))\n\nNumCorr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.363219Z","iopub.execute_input":"2026-02-06T16:42:43.363579Z","iopub.status.idle":"2026-02-06T16:42:43.532159Z","shell.execute_reply.started":"2026-02-06T16:42:43.363555Z","shell.execute_reply":"2026-02-06T16:42:43.531254Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### There is no linear relationship between the Target and these columns\n### NumCorr(corr()) ~= 0 ","metadata":{}},{"cell_type":"code","source":"\"\"\"We have a very large number of values, so the scatter is not clearly visible. Therefore, \nwe use this method to group the values into sets to reduce the large number, \nmaking the drawings appear clearer.\"\"\"\n\nq = pd.qcut(df['Annual Income'], q=20)\ngrouped = df.groupby(q)['Premium Amount'].mean()\n# >>>>>>>>>> 1 <<<<<<<<<<<<\ngrouped.plot(kind='line', marker='o')\nplt.title('Mean Premium Amount by Annual Income quantiles')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.533493Z","iopub.execute_input":"2026-02-06T16:42:43.533816Z","iopub.status.idle":"2026-02-06T16:42:43.861Z","shell.execute_reply.started":"2026-02-06T16:42:43.533787Z","shell.execute_reply":"2026-02-06T16:42:43.859828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# x = group for 'Annual Income'\ngrouped.index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.862102Z","iopub.execute_input":"2026-02-06T16:42:43.862486Z","iopub.status.idle":"2026-02-06T16:42:43.870672Z","shell.execute_reply.started":"2026-02-06T16:42:43.86246Z","shell.execute_reply":"2026-02-06T16:42:43.869602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q = pd.qcut(df['Annual Income'], q=100)\ngrouped = df.groupby(q)['Premium Amount'].mean()\n# >>>>>>>>>>>> 2 <<<<<<<<<<\nplt.figure(figsize=(8,5))\nplt.scatter(grouped.index.astype(str), grouped.values)\n\nplt.title('Mean Premium Amount by Annual Income quantiles')\nplt.xlabel('Annual Income Quantiles')\nplt.ylabel('Mean Premium Amount')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:43.871611Z","iopub.execute_input":"2026-02-06T16:42:43.872082Z","iopub.status.idle":"2026-02-06T16:42:44.732419Z","shell.execute_reply.started":"2026-02-06T16:42:43.872046Z","shell.execute_reply":"2026-02-06T16:42:44.731446Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **step-5 Outliers**","metadata":{}},{"cell_type":"markdown","source":"## first method >> Visulizatoin","metadata":{}},{"cell_type":"code","source":"fig=px.box(df,y='Age',title='outliers for Age')\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:44.733288Z","iopub.execute_input":"2026-02-06T16:42:44.733609Z","iopub.status.idle":"2026-02-06T16:42:45.015168Z","shell.execute_reply.started":"2026-02-06T16:42:44.733588Z","shell.execute_reply":"2026-02-06T16:42:45.013284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.016497Z","iopub.execute_input":"2026-02-06T16:42:45.017016Z","iopub.status.idle":"2026-02-06T16:42:45.026799Z","shell.execute_reply.started":"2026-02-06T16:42:45.016971Z","shell.execute_reply":"2026-02-06T16:42:45.022811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"for i in num:\n    fig=px.box(df,y=i,title=f'outliers for {i}')\n    fig.show()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.0284Z","iopub.execute_input":"2026-02-06T16:42:45.028817Z","iopub.status.idle":"2026-02-06T16:42:45.04905Z","shell.execute_reply.started":"2026-02-06T16:42:45.02879Z","shell.execute_reply":"2026-02-06T16:42:45.04738Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### maybe there are outliers in 'Annual Incomee'","metadata":{}},{"cell_type":"code","source":"outlier = df[df['Annual Income']< 1000]\n\noutlier.groupby('Occupation')['Annual Income'].agg(['mean','max','min'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.050431Z","iopub.execute_input":"2026-02-06T16:42:45.050869Z","iopub.status.idle":"2026-02-06T16:42:45.093169Z","shell.execute_reply.started":"2026-02-06T16:42:45.050791Z","shell.execute_reply":"2026-02-06T16:42:45.091661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1 =df['Annual Income'].quantile(0.25)\nq3 =df['Annual Income'].quantile(0.75)\nIQR = q3 - q1\nlower = q1 - 1.5*IQR\nupper = q3 + 1.5*IQR\noutliers = df[(df['Annual Income'] < lower) | (df['Annual Income'] > upper)]\noutliers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.094479Z","iopub.execute_input":"2026-02-06T16:42:45.094829Z","iopub.status.idle":"2026-02-06T16:42:45.193633Z","shell.execute_reply.started":"2026-02-06T16:42:45.094798Z","shell.execute_reply":"2026-02-06T16:42:45.192553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"A better and more comprehensive understanding of the nature of the data is necessary \nto determine whether the values ​​are outliers or not.\"\"\"\n\"\"\"for i in num:\n    q1 =A better and more comprehensive understanding of the nature of the data is necessary to determine whether the values ​​are outliers or not.df[i].quantile(0.25)\n    q3 =df[i].quantile(0.75)\n    IQR = q3 - q1\n    lower = q1 - 1.5*IQR\n    upper = q3 + 1.5*IQR\n    outliers = df[(df[i] < lower) | (df[i] > upper)]\noutliers\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.194695Z","iopub.execute_input":"2026-02-06T16:42:45.19522Z","iopub.status.idle":"2026-02-06T16:42:45.202479Z","shell.execute_reply.started":"2026-02-06T16:42:45.195185Z","shell.execute_reply":"2026-02-06T16:42:45.201429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.203904Z","iopub.execute_input":"2026-02-06T16:42:45.20427Z","iopub.status.idle":"2026-02-06T16:42:45.22931Z","shell.execute_reply.started":"2026-02-06T16:42:45.204236Z","shell.execute_reply":"2026-02-06T16:42:45.228234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T16:42:45.230609Z","iopub.execute_input":"2026-02-06T16:42:45.230986Z","iopub.status.idle":"2026-02-06T16:42:45.252467Z","shell.execute_reply.started":"2026-02-06T16:42:45.230957Z","shell.execute_reply":"2026-02-06T16:42:45.251128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Model Regression**","metadata":{}},{"cell_type":"code","source":"# convert date columen to year , month and day columens \ndf['year']=df['Policy Start Date'].dt.year\ndf['month']=df['Policy Start Date'].dt.month\ndf['day'] = df['Policy Start Date'].dt.day\ndf = df.drop('Policy Start Date',axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:04:20.790768Z","iopub.execute_input":"2026-02-06T17:04:20.791534Z","iopub.status.idle":"2026-02-06T17:04:21.098026Z","shell.execute_reply.started":"2026-02-06T17:04:20.791495Z","shell.execute_reply":"2026-02-06T17:04:21.096918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ENCODEING\nle = LabelEncoder()\nfor i in cat:\n    df[i]=le.fit_transform(df[i])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:03:44.350633Z","iopub.execute_input":"2026-02-06T17:03:44.351013Z","iopub.status.idle":"2026-02-06T17:03:45.157478Z","shell.execute_reply.started":"2026-02-06T17:03:44.350941Z","shell.execute_reply":"2026-02-06T17:03:45.156403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:04:26.90613Z","iopub.execute_input":"2026-02-06T17:04:26.90649Z","iopub.status.idle":"2026-02-06T17:04:27.516097Z","shell.execute_reply.started":"2026-02-06T17:04:26.906466Z","shell.execute_reply":"2026-02-06T17:04:27.515042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:03:56.005598Z","iopub.execute_input":"2026-02-06T17:03:56.00599Z","iopub.status.idle":"2026-02-06T17:03:56.082848Z","shell.execute_reply.started":"2026-02-06T17:03:56.005965Z","shell.execute_reply":"2026-02-06T17:03:56.081868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Splitting Data\nX = df.drop('Premium Amount',axis = 1)\nY = df['Premium Amount']\nx_train, x_test, y_train, y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:06:37.035212Z","iopub.execute_input":"2026-02-06T17:06:37.03613Z","iopub.status.idle":"2026-02-06T17:06:37.699033Z","shell.execute_reply.started":"2026-02-06T17:06:37.0361Z","shell.execute_reply":"2026-02-06T17:06:37.697595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:06:40.915064Z","iopub.execute_input":"2026-02-06T17:06:40.915523Z","iopub.status.idle":"2026-02-06T17:06:41.454545Z","shell.execute_reply.started":"2026-02-06T17:06:40.915497Z","shell.execute_reply":"2026-02-06T17:06:41.453506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:06:28.776209Z","iopub.execute_input":"2026-02-06T17:06:28.776554Z","iopub.status.idle":"2026-02-06T17:06:28.786082Z","shell.execute_reply.started":"2026-02-06T17:06:28.776533Z","shell.execute_reply":"2026-02-06T17:06:28.784869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"df['year']=df['Policy Start Date'].dt.year\ndf['month']=df[' Policy Start Date'].dt.month\ndf['day'] = df[' Policy Start Date'].dt.day\"\"\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# scalling Data\n\nscaler = StandardScaler()\n\nx_train_scaled = scaler.fit_transform(x_train)\nx_test_scaled = scaler.transform(x_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:07:12.010207Z","iopub.execute_input":"2026-02-06T17:07:12.010572Z","iopub.status.idle":"2026-02-06T17:07:12.522483Z","shell.execute_reply.started":"2026-02-06T17:07:12.010549Z","shell.execute_reply":"2026-02-06T17:07:12.521087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nmodel = LinearRegression()\nmodel.fit(x_train_scaled, y_train)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:07:15.480465Z","iopub.execute_input":"2026-02-06T17:07:15.480852Z","iopub.status.idle":"2026-02-06T17:07:17.012614Z","shell.execute_reply.started":"2026-02-06T17:07:15.480827Z","shell.execute_reply":"2026-02-06T17:07:17.011713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(x_test)\n\n\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(\"MAE:\", mae)\nprint(\"R2 Score:\", r2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T17:07:26.496816Z","iopub.execute_input":"2026-02-06T17:07:26.49717Z","iopub.status.idle":"2026-02-06T17:07:26.530864Z","shell.execute_reply.started":"2026-02-06T17:07:26.497143Z","shell.execute_reply":"2026-02-06T17:07:26.529886Z"}},"outputs":[],"execution_count":null}]}