{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:05:51.430468Z","iopub.execute_input":"2024-12-27T04:05:51.430863Z","iopub.status.idle":"2024-12-27T04:05:52.753274Z","shell.execute_reply.started":"2024-12-27T04:05:51.430816Z","shell.execute_reply":"2024-12-27T04:05:52.752074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#importing necessary libary\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:05:52.756092Z","iopub.execute_input":"2024-12-27T04:05:52.757187Z","iopub.status.idle":"2024-12-27T04:05:54.687881Z","shell.execute_reply.started":"2024-12-27T04:05:52.757136Z","shell.execute_reply":"2024-12-27T04:05:54.686504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_data=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsubmission=pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:05:54.690048Z","iopub.execute_input":"2024-12-27T04:05:54.690553Z","iopub.status.idle":"2024-12-27T04:06:05.128547Z","shell.execute_reply.started":"2024-12-27T04:05:54.690506Z","shell.execute_reply":"2024-12-27T04:06:05.127372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:05.12989Z","iopub.execute_input":"2024-12-27T04:06:05.130257Z","iopub.status.idle":"2024-12-27T04:06:05.173605Z","shell.execute_reply.started":"2024-12-27T04:06:05.130226Z","shell.execute_reply":"2024-12-27T04:06:05.172502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:05.175228Z","iopub.execute_input":"2024-12-27T04:06:05.175569Z","iopub.status.idle":"2024-12-27T04:06:05.18242Z","shell.execute_reply.started":"2024-12-27T04:06:05.175527Z","shell.execute_reply":"2024-12-27T04:06:05.181409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_with_na=[feature for feature in train_data.columns if train_data[feature].isnull().sum()>1]\nfeatures_with_na","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:05.183754Z","iopub.execute_input":"2024-12-27T04:06:05.184207Z","iopub.status.idle":"2024-12-27T04:06:05.797053Z","shell.execute_reply.started":"2024-12-27T04:06:05.184158Z","shell.execute_reply":"2024-12-27T04:06:05.795727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rows=int(np.ceil(len(features_with_na)/3))\ncol=3\n#creating figure and set size\nfig, axes=plt.subplots(rows,col,figsize=(15, rows*5))\naxes=axes.flatten()\n\n#looping throught features and creating suplots\nfor idx, feature in enumerate(features_with_na):\n    data=train_data.copy()\n    data[feature]=np.where(data[feature].isnull(),1,0)\n    #plot on current subplot\n    data.groupby(feature)['Premium Amount'].median().plot.bar(ax=axes[idx],color=['lightblue','orange'])\n    axes[idx].set_title(feature)\n    axes[idx].set_xlabel('') \n#hides any unused subplots\nfor idx in range(len(features_with_na), len(axes)):\n    fig.delaxes(axes[idx])\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:05.799432Z","iopub.execute_input":"2024-12-27T04:06:05.799772Z","iopub.status.idle":"2024-12-27T04:06:10.076429Z","shell.execute_reply.started":"2024-12-27T04:06:05.79974Z","shell.execute_reply":"2024-12-27T04:06:10.075219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"By above figure it is noticeable that values that are null(in orange color) have significant impact on our dependent variable","metadata":{}},{"cell_type":"code","source":"numerical_features=[feature for feature in train_data.columns if train_data[feature].dtype!='O']\ntrain_data[numerical_features].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:10.077628Z","iopub.execute_input":"2024-12-27T04:06:10.077946Z","iopub.status.idle":"2024-12-27T04:06:10.130106Z","shell.execute_reply.started":"2024-12-27T04:06:10.077894Z","shell.execute_reply":"2024-12-27T04:06:10.128833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"discrete_features=[feature for feature in numerical_features if len(train_data[feature].unique())<25 and feature not in ['Id']]\nprint('Discrete features count {}'.format(len(discrete_features)))\n#making subplots\nrows=int(np.ceil(len(discrete_features)/2))\ncol=2\n#creating figure and set size\nfig, axes=plt.subplots(rows,col,figsize=(15, rows*5))\naxes=axes.flatten()\n\n#looping throught features and creating suplots\nfor idx, feature in enumerate(discrete_features):\n    data=train_data.copy()\n    #plot on current subplot\n    data.groupby(feature)['Premium Amount'].median().plot.bar(ax=axes[idx])\n    axes[idx].set_title(feature)\n    axes[idx].set_xlabel('') \n#hides any unused subplots\nfor idx in range(len(features_with_na), len(axes)):\n    fig.delaxes(axes[idx])\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:10.131559Z","iopub.execute_input":"2024-12-27T04:06:10.132019Z","iopub.status.idle":"2024-12-27T04:06:12.443239Z","shell.execute_reply.started":"2024-12-27T04:06:10.131972Z","shell.execute_reply":"2024-12-27T04:06:12.442008Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We can see the relation between discrete numerical features and dependent variable","metadata":{}},{"cell_type":"code","source":"continuous_features=[feature for feature in numerical_features if feature not in discrete_features+['id'] ]\nprint('Continuous features count {}'.format(len(continuous_features)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:12.444685Z","iopub.execute_input":"2024-12-27T04:06:12.445158Z","iopub.status.idle":"2024-12-27T04:06:12.452008Z","shell.execute_reply.started":"2024-12-27T04:06:12.445113Z","shell.execute_reply":"2024-12-27T04:06:12.450925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for feature in continuous_features:\n    data=train_data.copy()\n    data[feature].hist(bins=25)\n    plt.xlabel(feature)\n    plt.ylabel(\"Count\")\n    plt.title(feature)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T04:06:12.453368Z","iopub.execute_input":"2024-12-27T04:06:12.453752Z","iopub.status.idle":"2024-12-27T04:06:14.608064Z","shell.execute_reply.started":"2024-12-27T04:06:12.453722Z","shell.execute_reply":"2024-12-27T04:06:14.606938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Annual income and premium are skewed and have outliers","metadata":{}}]}