{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:22.609359Z","iopub.execute_input":"2024-12-01T10:08:22.609869Z","iopub.status.idle":"2024-12-01T10:08:22.619373Z","shell.execute_reply.started":"2024-12-01T10:08:22.609817Z","shell.execute_reply":"2024-12-01T10:08:22.618194Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *This is an starter EDA notebook for your reference.* \n# *If there are any issues, kindly comment it down and I'll take a note of it.*","metadata":{}},{"cell_type":"markdown","source":"*Importing libraries*","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:22.621668Z","iopub.execute_input":"2024-12-01T10:08:22.622222Z","iopub.status.idle":"2024-12-01T10:08:22.638669Z","shell.execute_reply.started":"2024-12-01T10:08:22.622169Z","shell.execute_reply":"2024-12-01T10:08:22.637108Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*The competition has a dataset of its own and anither dataset has been given for reference in the data section. I have imported both of them and concatenated them to make more data as they contain same type of values*","metadata":{}},{"cell_type":"code","source":"df1 = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:22.640977Z","iopub.execute_input":"2024-12-01T10:08:22.641531Z","iopub.status.idle":"2024-12-01T10:08:27.33594Z","shell.execute_reply.started":"2024-12-01T10:08:22.64148Z","shell.execute_reply":"2024-12-01T10:08:27.334928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:27.337104Z","iopub.execute_input":"2024-12-01T10:08:27.337387Z","iopub.status.idle":"2024-12-01T10:08:27.362218Z","shell.execute_reply.started":"2024-12-01T10:08:27.337359Z","shell.execute_reply":"2024-12-01T10:08:27.361045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:27.365181Z","iopub.execute_input":"2024-12-01T10:08:27.365556Z","iopub.status.idle":"2024-12-01T10:08:27.375894Z","shell.execute_reply.started":"2024-12-01T10:08:27.365517Z","shell.execute_reply":"2024-12-01T10:08:27.374833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2 = pd.read_csv('/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:27.377313Z","iopub.execute_input":"2024-12-01T10:08:27.377802Z","iopub.status.idle":"2024-12-01T10:08:28.358367Z","shell.execute_reply.started":"2024-12-01T10:08:27.377661Z","shell.execute_reply":"2024-12-01T10:08:28.357285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.359723Z","iopub.execute_input":"2024-12-01T10:08:28.36015Z","iopub.status.idle":"2024-12-01T10:08:28.382631Z","shell.execute_reply.started":"2024-12-01T10:08:28.360094Z","shell.execute_reply":"2024-12-01T10:08:28.381513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.384074Z","iopub.execute_input":"2024-12-01T10:08:28.384506Z","iopub.status.idle":"2024-12-01T10:08:28.393561Z","shell.execute_reply.started":"2024-12-01T10:08:28.384439Z","shell.execute_reply":"2024-12-01T10:08:28.392526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.394781Z","iopub.execute_input":"2024-12-01T10:08:28.395425Z","iopub.status.idle":"2024-12-01T10:08:28.411207Z","shell.execute_reply.started":"2024-12-01T10:08:28.395393Z","shell.execute_reply":"2024-12-01T10:08:28.409997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.412989Z","iopub.execute_input":"2024-12-01T10:08:28.41341Z","iopub.status.idle":"2024-12-01T10:08:28.426106Z","shell.execute_reply.started":"2024-12-01T10:08:28.413362Z","shell.execute_reply":"2024-12-01T10:08:28.425042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Example: Assuming df1 and df2 are your DataFrames\n\n# Function to ensure the target column is the last column\ndef move_target_to_end(df, target_column):\n    cols = [col for col in df.columns if col != target_column]  # All columns except the target\n    cols.append(target_column)  # Add the target column to the end\n    return df[cols]\n\n# Target column name\ntarget_column = \"Premium Amount\"\n\n# Reorder columns for both DataFrames\ndf1 = move_target_to_end(df1, target_column)\ndf2 = move_target_to_end(df2, target_column)\n\n# Concatenate the DataFrames\nresult = pd.concat([df1, df2], ignore_index=True)\n\n# Display the result\nresult.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.429748Z","iopub.execute_input":"2024-12-01T10:08:28.430113Z","iopub.status.idle":"2024-12-01T10:08:28.920044Z","shell.execute_reply.started":"2024-12-01T10:08:28.430064Z","shell.execute_reply":"2024-12-01T10:08:28.918723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Creating the training dataset*","metadata":{}},{"cell_type":"code","source":"train = result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.921244Z","iopub.execute_input":"2024-12-01T10:08:28.921596Z","iopub.status.idle":"2024-12-01T10:08:28.926685Z","shell.execute_reply.started":"2024-12-01T10:08:28.921564Z","shell.execute_reply":"2024-12-01T10:08:28.925526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Shape format:- (Number of Rows x Number of Columns)*","metadata":{}},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.92825Z","iopub.execute_input":"2024-12-01T10:08:28.928667Z","iopub.status.idle":"2024-12-01T10:08:28.940337Z","shell.execute_reply.started":"2024-12-01T10:08:28.928612Z","shell.execute_reply":"2024-12-01T10:08:28.938745Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Dropping the id column as it is of no use*","metadata":{}},{"cell_type":"code","source":"train = train.drop(columns = 'id',axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:28.941902Z","iopub.execute_input":"2024-12-01T10:08:28.942235Z","iopub.status.idle":"2024-12-01T10:08:29.208719Z","shell.execute_reply.started":"2024-12-01T10:08:28.942203Z","shell.execute_reply":"2024-12-01T10:08:29.207668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Checking for null values*","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:29.210061Z","iopub.execute_input":"2024-12-01T10:08:29.210456Z","iopub.status.idle":"2024-12-01T10:08:29.993861Z","shell.execute_reply.started":"2024-12-01T10:08:29.210411Z","shell.execute_reply":"2024-12-01T10:08:29.99283Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*What are the data types in this dataset?*","metadata":{}},{"cell_type":"code","source":"train.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:29.995067Z","iopub.execute_input":"2024-12-01T10:08:29.995367Z","iopub.status.idle":"2024-12-01T10:08:30.003125Z","shell.execute_reply.started":"2024-12-01T10:08:29.995338Z","shell.execute_reply":"2024-12-01T10:08:30.002013Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Checking for any duplicate values*","metadata":{}},{"cell_type":"code","source":"duplicated = train.duplicated()\nprint(train[duplicated])\nprint(f\"Number of duplicate rows: {duplicated.sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:30.00458Z","iopub.execute_input":"2024-12-01T10:08:30.005002Z","iopub.status.idle":"2024-12-01T10:08:32.471294Z","shell.execute_reply.started":"2024-12-01T10:08:30.004958Z","shell.execute_reply":"2024-12-01T10:08:32.470013Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*I wanted to check if there were any entries with \"Age\" being under 18 as people don't own bank accounts if thats the case.*","metadata":{}},{"cell_type":"code","source":"train['Age'].min()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:32.472431Z","iopub.execute_input":"2024-12-01T10:08:32.472791Z","iopub.status.idle":"2024-12-01T10:08:32.493314Z","shell.execute_reply.started":"2024-12-01T10:08:32.472756Z","shell.execute_reply":"2024-12-01T10:08:32.492175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *Now, a series of box-plots are shown below. They contain relations between all numerical columns.*","metadata":{}},{"cell_type":"markdown","source":"*Trend of Annual Income vs Age*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Age', y='Annual Income', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Annual Income vs Age\", fontsize=16)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Annual Income\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:49:15.096228Z","iopub.execute_input":"2024-12-01T10:49:15.096724Z","iopub.status.idle":"2024-12-01T10:49:16.402772Z","shell.execute_reply.started":"2024-12-01T10:49:15.096685Z","shell.execute_reply":"2024-12-01T10:49:16.401732Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Credit Score vs Insurance Duration*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Insurance Duration', y='Credit Score', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Credit Score vs Insurance Duration\", fontsize=16)\nplt.xlabel(\"Insurance Duration\", fontsize=12)\nplt.ylabel(\"Credit Score\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:48:32.965943Z","iopub.execute_input":"2024-12-01T10:48:32.966306Z","iopub.status.idle":"2024-12-01T10:48:33.528668Z","shell.execute_reply.started":"2024-12-01T10:48:32.966274Z","shell.execute_reply":"2024-12-01T10:48:33.527491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Number of Dependents vs Age*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Age', y='Number of Dependents', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Number of Dependents vs Age\", fontsize=16)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Number of Dependents\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:49:51.016066Z","iopub.execute_input":"2024-12-01T10:49:51.016409Z","iopub.status.idle":"2024-12-01T10:49:52.187323Z","shell.execute_reply.started":"2024-12-01T10:49:51.01638Z","shell.execute_reply":"2024-12-01T10:49:52.186137Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Health Score vs Age*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Age', y='Health Score', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Health Score vs Age\", fontsize=16)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Health Score\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:50:25.809071Z","iopub.execute_input":"2024-12-01T10:50:25.809483Z","iopub.status.idle":"2024-12-01T10:50:26.936844Z","shell.execute_reply.started":"2024-12-01T10:50:25.809426Z","shell.execute_reply":"2024-12-01T10:50:26.935733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Credit Score vs Age*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Age', y='Credit Score', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Credit Score vs Age\", fontsize=16)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Credit Score\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:50:59.203609Z","iopub.execute_input":"2024-12-01T10:50:59.204012Z","iopub.status.idle":"2024-12-01T10:51:00.308583Z","shell.execute_reply.started":"2024-12-01T10:50:59.203976Z","shell.execute_reply":"2024-12-01T10:51:00.307449Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Credit Score vs Vehicle Age*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Vehicle Age', y='Credit Score', data=train)\n\n# Set plot labels and title\nplt.title(\"Vehicle Age vs Credit Score\", fontsize=16)\nplt.xlabel(\"Vehicle Age\", fontsize=12)\nplt.ylabel(\"Credit Score\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:52:55.099414Z","iopub.execute_input":"2024-12-01T10:52:55.099882Z","iopub.status.idle":"2024-12-01T10:52:59.902112Z","shell.execute_reply.started":"2024-12-01T10:52:55.099843Z","shell.execute_reply":"2024-12-01T10:52:59.900944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Premium Amount vs Age*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Age', y='Premium Amount', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Premium Amount vs Age\", fontsize=16)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Premium Amount\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:56:16.758454Z","iopub.execute_input":"2024-12-01T10:56:16.758844Z","iopub.status.idle":"2024-12-01T10:56:17.888354Z","shell.execute_reply.started":"2024-12-01T10:56:16.758812Z","shell.execute_reply":"2024-12-01T10:56:17.887172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Premium Amount vs Number of Dependents*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Number of Dependents', y='Premium Amount', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Premium Amount vs Number of Dependents\", fontsize=16)\nplt.xlabel(\"Number of Dependents\", fontsize=12)\nplt.ylabel(\"Premium Amount\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:59:47.169991Z","iopub.execute_input":"2024-12-01T10:59:47.170376Z","iopub.status.idle":"2024-12-01T10:59:47.701288Z","shell.execute_reply.started":"2024-12-01T10:59:47.170338Z","shell.execute_reply":"2024-12-01T10:59:47.699936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"*Trend of Premium Amount vs Insurance Duration*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='Insurance Duration', y='Premium Amount', data=train)\n\n# Set plot labels and title\nplt.title(\"Boxplot of Premium Amount vs Insurance Duration\", fontsize=16)\nplt.xlabel(\"Insurance Duration\", fontsize=12)\nplt.ylabel(\"Premium Amount\", fontsize=12)\nplt.xticks(rotation=45)  # Rotate x-axis labels if necessary\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:01:44.487456Z","iopub.execute_input":"2024-12-01T11:01:44.4884Z","iopub.status.idle":"2024-12-01T11:01:45.079113Z","shell.execute_reply.started":"2024-12-01T11:01:44.488361Z","shell.execute_reply":"2024-12-01T11:01:45.078015Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *Below is the distribution of numerical columns to check for normal distribution and transformations.* \n# *If right-skewed, log transformation will work, if left-skewed, log transformation will not work.* ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Create histograms for numerical columns\nplt.figure(figsize=(12, 8))\ntrain[['Age', 'Annual Income', 'Health Score', 'Vehicle Age', 'Credit Score', 'Insurance Duration']].hist(bins=20, figsize=(12, 8), edgecolor='black')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:33.748777Z","iopub.execute_input":"2024-12-01T10:08:33.749109Z","iopub.status.idle":"2024-12-01T10:08:35.317665Z","shell.execute_reply.started":"2024-12-01T10:08:33.749078Z","shell.execute_reply":"2024-12-01T10:08:35.316546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *For categorical columns, we will see if there is any class imbalance.*\n# *Class imbalance can be fixed in two ways:- Oversampling and Undersampling.*\n# *Using SMOTE is a very famous way of handling class imbalance.*","metadata":{}},{"cell_type":"code","source":"# Create countplots for categorical columns\nplt.figure(figsize=(10, 6))\nsns.countplot(x='Gender', data=train)\nplt.title(\"Distribution of Gender\", fontsize=16)\nplt.show()\n\nplt.figure(figsize=(10, 6))\nsns.countplot(x='Marital Status', data=train)\nplt.title(\"Distribution of Marital Status\", fontsize=16)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:08:35.319033Z","iopub.execute_input":"2024-12-01T10:08:35.319412Z","iopub.status.idle":"2024-12-01T10:08:37.240719Z","shell.execute_reply.started":"2024-12-01T10:08:35.319371Z","shell.execute_reply":"2024-12-01T10:08:37.239639Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *Now, a correlation matrix between the numerical features and target variable*","metadata":{}},{"cell_type":"code","source":"temp = train[['Age', 'Annual Income', 'Health Score', 'Vehicle Age', 'Credit Score', 'Insurance Duration','Premium Amount']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:04:32.962235Z","iopub.execute_input":"2024-12-01T11:04:32.962616Z","iopub.status.idle":"2024-12-01T11:04:33.007896Z","shell.execute_reply.started":"2024-12-01T11:04:32.962582Z","shell.execute_reply":"2024-12-01T11:04:33.00675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(temp.corr(),annot = True, fmt = \"0.2f\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:04:34.858695Z","iopub.execute_input":"2024-12-01T11:04:34.859078Z","iopub.status.idle":"2024-12-01T11:04:35.590326Z","shell.execute_reply.started":"2024-12-01T11:04:34.859044Z","shell.execute_reply":"2024-12-01T11:04:35.589167Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *I hope it is useful to you. There would be more EDA ways but I could only think of these ones. Do tell me more in the comments.*\n# *By visualizing these trends, you would be able to impute and transform the data as per your convenience. I haven't performed any imputation on my own as it might interfere with the actual trends*","metadata":{}},{"cell_type":"markdown","source":"# *If you liked it, please give it an upvote.*\n# *Thank you!*","metadata":{}}]}