{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:41.35457Z","iopub.execute_input":"2025-07-12T14:23:41.354933Z","iopub.status.idle":"2025-07-12T14:23:41.375506Z","shell.execute_reply.started":"2025-07-12T14:23:41.354909Z","shell.execute_reply":"2025-07-12T14:23:41.374334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt \nimport plotly.express as xp\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:41.37702Z","iopub.execute_input":"2025-07-12T14:23:41.377371Z","iopub.status.idle":"2025-07-12T14:23:41.383348Z","shell.execute_reply.started":"2025-07-12T14:23:41.37734Z","shell.execute_reply":"2025-07-12T14:23:41.382526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:41.384345Z","iopub.execute_input":"2025-07-12T14:23:41.384737Z","iopub.status.idle":"2025-07-12T14:23:51.78656Z","shell.execute_reply.started":"2025-07-12T14:23:41.384704Z","shell.execute_reply":"2025-07-12T14:23:51.785624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:51.788685Z","iopub.execute_input":"2025-07-12T14:23:51.788976Z","iopub.status.idle":"2025-07-12T14:23:52.513711Z","shell.execute_reply.started":"2025-07-12T14:23:51.788955Z","shell.execute_reply":"2025-07-12T14:23:52.51285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA : exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:52.514668Z","iopub.execute_input":"2025-07-12T14:23:52.514978Z","iopub.status.idle":"2025-07-12T14:23:53.228047Z","shell.execute_reply.started":"2025-07-12T14:23:52.514952Z","shell.execute_reply":"2025-07-12T14:23:53.227004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1- Handling Missing Values","metadata":{}},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:53.22885Z","iopub.execute_input":"2025-07-12T14:23:53.229155Z","iopub.status.idle":"2025-07-12T14:23:53.928982Z","shell.execute_reply.started":"2025-07-12T14:23:53.229135Z","shell.execute_reply":"2025-07-12T14:23:53.928146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# missing percentage\nmissings = train_data.isnull().mean()*100\nmissings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:53.929985Z","iopub.execute_input":"2025-07-12T14:23:53.930682Z","iopub.status.idle":"2025-07-12T14:23:54.622429Z","shell.execute_reply.started":"2025-07-12T14:23:53.930653Z","shell.execute_reply":"2025-07-12T14:23:54.62163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# handling Age ( mean, mode, median)\nsns.boxplot(x=train_data['Age'])\nplt.title(\"BoxPlot of Age\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:54.623299Z","iopub.execute_input":"2025-07-12T14:23:54.624021Z","iopub.status.idle":"2025-07-12T14:23:54.817989Z","shell.execute_reply.started":"2025-07-12T14:23:54.623998Z","shell.execute_reply":"2025-07-12T14:23:54.816962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data['Age'].mean())\nprint(train_data['Age'].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:54.818846Z","iopub.execute_input":"2025-07-12T14:23:54.819151Z","iopub.status.idle":"2025-07-12T14:23:54.852508Z","shell.execute_reply.started":"2025-07-12T14:23:54.819122Z","shell.execute_reply":"2025-07-12T14:23:54.851627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:54.855657Z","iopub.execute_input":"2025-07-12T14:23:54.855922Z","iopub.status.idle":"2025-07-12T14:23:55.498344Z","shell.execute_reply.started":"2025-07-12T14:23:54.855901Z","shell.execute_reply":"2025-07-12T14:23:55.497522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in train_data.columns:\n    print(train_data[col].value_counts(normalize=True))\n    print(\"-------------\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:55.499161Z","iopub.execute_input":"2025-07-12T14:23:55.499426Z","iopub.status.idle":"2025-07-12T14:23:57.242549Z","shell.execute_reply.started":"2025-07-12T14:23:55.499407Z","shell.execute_reply":"2025-07-12T14:23:57.241646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:57.243338Z","iopub.execute_input":"2025-07-12T14:23:57.243652Z","iopub.status.idle":"2025-07-12T14:23:57.947802Z","shell.execute_reply.started":"2025-07-12T14:23:57.243622Z","shell.execute_reply":"2025-07-12T14:23:57.946842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_data['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:57.948867Z","iopub.execute_input":"2025-07-12T14:23:57.949479Z","iopub.status.idle":"2025-07-12T14:23:57.961628Z","shell.execute_reply.started":"2025-07-12T14:23:57.949448Z","shell.execute_reply":"2025-07-12T14:23:57.960631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num = train_data.select_dtypes(include=['int64', 'float64']).columns\ncat = train_data.select_dtypes(include=['object', 'category']).columns\nprint(num.tolist())\nprint(cat.tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:57.962493Z","iopub.execute_input":"2025-07-12T14:23:57.962761Z","iopub.status.idle":"2025-07-12T14:23:58.647076Z","shell.execute_reply.started":"2025-07-12T14:23:57.962742Z","shell.execute_reply":"2025-07-12T14:23:58.646315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nnumerical_columns = ['Age' ,'Annual Income' ,'' ]\ncat \ntrain_data.fillna(train_data.mean(), inplace=True)\n#gender, Marital Status\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:58.648013Z","iopub.execute_input":"2025-07-12T14:23:58.648332Z","iopub.status.idle":"2025-07-12T14:23:58.654708Z","shell.execute_reply.started":"2025-07-12T14:23:58.648302Z","shell.execute_reply":"2025-07-12T14:23:58.653827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in num:\n    plt.figure(figsize=(15,8))\n    sns.histplot(x=train_data[col] , kde=True , bins = 25, data =train_data)\n    plt.title(f\"Histogram of {col}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:23:58.655745Z","iopub.execute_input":"2025-07-12T14:23:58.656041Z","iopub.status.idle":"2025-07-12T14:24:41.860765Z","shell.execute_reply.started":"2025-07-12T14:23:58.656016Z","shell.execute_reply":"2025-07-12T14:24:41.859813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_data.fillna(test_data.mean() , inplace=True )\n# missing percentage\nfor col in num:\n    print(f\"Column Name : {col}\")\n    print(train_data[col].isnull().mean()*100)\n    \n    train_data[col] = train_data[col].fillna(train_data[col].mean())\n    print(f\"Number of missing after Handling : {train_data[col].isnull().mean()*100}\")\n    print(\"________\")\n    \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:24:41.861803Z","iopub.execute_input":"2025-07-12T14:24:41.862062Z","iopub.status.idle":"2025-07-12T14:24:42.156671Z","shell.execute_reply.started":"2025-07-12T14:24:41.862041Z","shell.execute_reply":"2025-07-12T14:24:42.155762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in num:\n    plt.figure(figsize=(15,8))\n    sns.histplot(x=train_data[col] , kde=True , bins = 25 , data= train_data)\n    plt.title(f\"Histogram of {col}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:24:42.157532Z","iopub.execute_input":"2025-07-12T14:24:42.157856Z","iopub.status.idle":"2025-07-12T14:25:26.709839Z","shell.execute_reply.started":"2025-07-12T14:24:42.157834Z","shell.execute_reply":"2025-07-12T14:25:26.708952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Marital Status'].mode()[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:25:26.710833Z","iopub.execute_input":"2025-07-12T14:25:26.711121Z","iopub.status.idle":"2025-07-12T14:25:26.804935Z","shell.execute_reply.started":"2025-07-12T14:25:26.711098Z","shell.execute_reply":"2025-07-12T14:25:26.803989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cat:\n    print(f\"Column Name : {col}\")\n    print(train_data[col].isnull().mean()*100)\n    \n    train_data[col] = train_data[col].fillna(train_data[col].mode()[0])\n    print(f\"Number of missing after Handling : {train_data[col].isnull().mean()*100}\")\n    print(\"________\")\n    \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:25:26.805939Z","iopub.execute_input":"2025-07-12T14:25:26.806281Z","iopub.status.idle":"2025-07-12T14:25:30.335694Z","shell.execute_reply.started":"2025-07-12T14:25:26.806253Z","shell.execute_reply":"2025-07-12T14:25:30.334885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Marital Status - Occupation - Customer Feedback ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:25:30.336464Z","iopub.execute_input":"2025-07-12T14:25:30.336731Z","iopub.status.idle":"2025-07-12T14:25:30.340643Z","shell.execute_reply.started":"2025-07-12T14:25:30.336708Z","shell.execute_reply":"2025-07-12T14:25:30.339845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Occupation'] = train_data['Occupation'].fillna('Unknown')\ntrain_data['Customer Feedback'] = train_data['Customer Feedback'].fillna('Unknown')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:25:30.341467Z","iopub.execute_input":"2025-07-12T14:25:30.341773Z","iopub.status.idle":"2025-07-12T14:25:30.514186Z","shell.execute_reply.started":"2025-07-12T14:25:30.341749Z","shell.execute_reply":"2025-07-12T14:25:30.513385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### After considering Data distribution , we found out that filling these two colmns 'Number of Dependents' , 'Credit Score' with mean value would have a negative effective and cause the data to have too much outliers, so we need to handle it with a different way!\n","metadata":{}},{"cell_type":"code","source":"# 'Number of Dependents' , 'Credit Score'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:25:30.515001Z","iopub.execute_input":"2025-07-12T14:25:30.515233Z","iopub.status.idle":"2025-07-12T14:25:30.519139Z","shell.execute_reply.started":"2025-07-12T14:25:30.515215Z","shell.execute_reply":"2025-07-12T14:25:30.518348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle missing values for 'Number of Dependents' and 'Credit Score' using median\n\ntrain_data['Number of Dependents'] = train_data['Number of Dependents'].fillna(train_data['Number of Dependents'].median())\ntrain_data['Credit Score'] = train_data['Credit Score'].fillna(train_data['Credit Score'].median())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:25:30.520118Z","iopub.execute_input":"2025-07-12T14:25:30.52043Z","iopub.status.idle":"2025-07-12T14:25:30.637503Z","shell.execute_reply.started":"2025-07-12T14:25:30.5204Z","shell.execute_reply":"2025-07-12T14:25:30.636635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data['Number of Dependents'].isnull().sum())\nprint(train_data['Credit Score'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:21:29.640978Z","iopub.execute_input":"2025-07-12T14:21:29.641749Z","iopub.status.idle":"2025-07-12T14:21:29.659515Z","shell.execute_reply.started":"2025-07-12T14:21:29.641716Z","shell.execute_reply":"2025-07-12T14:21:29.658524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Missing values per column:\")\nprint(train_data.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:27:25.822476Z","iopub.execute_input":"2025-07-12T14:27:25.822837Z","iopub.status.idle":"2025-07-12T14:27:26.547639Z","shell.execute_reply.started":"2025-07-12T14:27:25.822813Z","shell.execute_reply":"2025-07-12T14:27:26.546783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# step 2 Detecting outliers ","metadata":{}},{"cell_type":"code","source":"#Outlier detection for numerical columns\n#Plot boxplots for each numerical column to visually spot outliers.\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nnum_cols = train_data.select_dtypes(include=['int64', 'float64']).columns\n\nfor col in num_cols:\n    plt.figure(figsize=(10, 5))\n    sns.boxplot(x=train_data[col])\n    plt.title(f'Boxplot of {col}')\n    plt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:56:20.317503Z","iopub.execute_input":"2025-07-12T14:56:20.317872Z","iopub.status.idle":"2025-07-12T14:56:23.000316Z","shell.execute_reply.started":"2025-07-12T14:56:20.317849Z","shell.execute_reply":"2025-07-12T14:56:22.999207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#handling outliers for each numerical columns\n\nnum_cols = train_data.select_dtypes(include=['int64', 'float64']).columns\n\nfor col in num_cols:\n    Q1 = train_data[col].quantile(0.25)\n    Q3 = train_data[col].quantile(0.75)\n    IQR = Q3 - Q1\n\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    # Print before clipping info\n    print(f\"{col}: Lower Bound = {lower_bound}, Upper Bound = {upper_bound}\")\n\n    train_data[col] = train_data[col].clip(lower=lower_bound, upper=upper_bound)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:10:31.791721Z","iopub.execute_input":"2025-07-12T15:10:31.792148Z","iopub.status.idle":"2025-07-12T15:10:32.546444Z","shell.execute_reply.started":"2025-07-12T15:10:31.792114Z","shell.execute_reply":"2025-07-12T15:10:32.545453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nnum_cols = train_data.select_dtypes(include=['int64', 'float64']).columns\n\nfor col in num_cols:\n    plt.figure(figsize=(10, 5))\n    sns.boxplot(x=train_data[col])\n    plt.title(f'Boxplot of {col} after outlier handling')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:13:00.265695Z","iopub.execute_input":"2025-07-12T15:13:00.26605Z","iopub.status.idle":"2025-07-12T15:13:02.195633Z","shell.execute_reply.started":"2025-07-12T15:13:00.266028Z","shell.execute_reply":"2025-07-12T15:13:02.194556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ask Ola about it why i did this , is this important to do it??\n#i had this idea using the internet because i want it to be sure that i did handle all the outliers.\nnum_cols = train_data.select_dtypes(include=['int64', 'float64']).columns\n\nfor col in num_cols:\n    Q1 = train_data[col].quantile(0.25)\n    Q3 = train_data[col].quantile(0.75)\n    IQR = Q3 - Q1\n    \n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    # Count outliers before clipping (if you kept a copy of original data)\n    # If you didn’t, run this before clipping next time.\n    \n    # Count outliers after clipping\n    outliers_after = train_data[(train_data[col] < lower_bound) | (train_data[col] > upper_bound)][col].count()\n    \n    print(f\"{col} outliers after handling: {outliers_after}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:13:48.422139Z","iopub.execute_input":"2025-07-12T15:13:48.422491Z","iopub.status.idle":"2025-07-12T15:13:49.041363Z","shell.execute_reply.started":"2025-07-12T15:13:48.422464Z","shell.execute_reply":"2025-07-12T15:13:49.039897Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Step 3: Encoding**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}