{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Getting the data","metadata":{}},{"cell_type":"code","source":"# !kaggle competitions download -c playground-series-s4e12\n\n# import zipfile\n# import os\n# import pandas as pd\n\n# with zipfile.ZipFile('playground-series-s4e12.zip', 'r') as zip_ref:\n#     zip_ref.extractall('.')\n\n# os.remove('playground-series-s4e12.zip')","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ntrain = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Baseline","metadata":{}},{"cell_type":"code","source":"train.dtypes","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isna().sum()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocessing\n\n🚀 The code utilizes various imputation strategies to address missing data in both training and test datasets:\n\nNumerical Features:\n\n📅 Age: Filled with the training set's mean age.\n\n💰 Annual Income: Imputed using the median annual income from the training set.\n\n👨‍👩‍👧 Number of Dependents: Replaced with the mean number of dependents from the training set.\n\n🏥 Health Score: Filled with the mean health score from the training set.\n\n📄 Previous Claims: Imputed using the median number of previous claims from the training set.\n\n🚗 Vehicle Age: Replaced with the mean vehicle age from the training set.\n\n💳 Credit Score: Filled with the mean credit score from the training set.\n\n⏳ Insurance Duration: Imputed using the mean insurance duration from the training set.\n\nCategorical Features:\n\n💍 Marital Status: Missing values are randomly assigned as 'Single', 'Married', or 'Divorced'.\n\n👔 Occupation: Missing entries are labeled as 'Missing'.\n\n⭐ Customer Feedback: Missing values are randomly assigned as 'Poor', 'Average', or 'Good'.\n\nFeature Removal:\n\n🆔 'id' and 'Policy Start Date': These columns are dropped from both datasets, likely because they are unique identifiers or not directly relevant to the analysis.\n\n🎯 These imputation methods assume:\n\nThe training set's statistical measures (mean or median) are appropriate substitutes for missing values.\nCategorical missingness can be addressed by random assignment or placeholder labels.","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\nmean_age = train['Age'].mean()\ntrain['Age'] = train['Age'].fillna(mean_age)\ntest['Age'] = test['Age'].fillna(mean_age)\n\nmedian_annual_income = train['Annual Income'].median()\ntrain['Annual Income'] = train['Annual Income'].fillna(median_annual_income)\ntest['Annual Income'] = test['Annual Income'].fillna(median_annual_income)\n\ntrain['Marital Status'] = train['Marital Status'].fillna(np.random.choice(['Single', 'Married', 'Divorced']))\ntest['Marital Status'] = test['Marital Status'].fillna(np.random.choice(['Single', 'Married', 'Divorced']))\n\nmean_number_dependents = train['Number of Dependents'].mean()\ntrain['Number of Dependents'] = train['Number of Dependents'].fillna(mean_number_dependents)\ntest['Number of Dependents'] = test['Number of Dependents'].fillna(mean_number_dependents)\n\ntrain['Occupation'] = train['Occupation'].fillna('Missing')\ntest['Occupation'] = test['Occupation'].fillna('Missing')\n\nmean_health_score = train['Health Score'].mean()\ntrain['Health Score'] = train['Health Score'].fillna(mean_health_score)\ntest['Health Score'] = test['Health Score'].fillna(mean_health_score)\n\nmedian_previous_claims = train['Previous Claims'].median()\ntrain['Previous Claims'] = train['Previous Claims'].fillna(median_previous_claims)\ntest['Previous Claims'] = test['Previous Claims'].fillna(median_previous_claims)\n\nmean_vehicle_age = train['Vehicle Age'].mean()\ntrain['Vehicle Age'] = train['Vehicle Age'].fillna(mean_vehicle_age)\ntest['Vehicle Age'] = test['Vehicle Age'].fillna(mean_vehicle_age)\n\nmean_credit_score = train['Credit Score'].mean()\ntrain['Credit Score'] = train['Credit Score'].fillna(mean_credit_score)\ntest['Credit Score'] = test['Credit Score'].fillna(mean_credit_score)\n\nmean_insurance_duration = train['Insurance Duration'].mean()\ntrain['Insurance Duration'] = train['Insurance Duration'].fillna(mean_insurance_duration)\ntest['Insurance Duration'] = test['Insurance Duration'].fillna(mean_insurance_duration)\n\ntrain['Customer Feedback'] = train['Customer Feedback'].fillna(np.random.choice(['Poor', 'Average', 'Good']))\ntest['Customer Feedback'] = test['Customer Feedback'].fillna(np.random.choice(['Poor', 'Average', 'Good']))\n\ntrain.drop(columns =['id', 'Policy Start Date'], inplace=True)\ntest.drop(columns =['id','Policy Start Date'], inplace=True)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# all object columns to categorical\nfor col in train.select_dtypes('object').columns:\n    train[col] = train[col].astype('category')\n    test[col] = test[col].astype('category')","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform One Hot Encoding on the categorical columns\ntrain = pd.get_dummies(train, columns=train.select_dtypes('category').columns, drop_first=True)\ntest = pd.get_dummies(test, columns=test.select_dtypes('category').columns, drop_first=True)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop(columns='Premium Amount', axis=1)\ny = train['Premium Amount']","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standardize all the data with values from the training set\nfrom sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\n\n# Align the test set columns to the training set columns\ntest = test.reindex(columns=X.columns)\n\n# Select columns that are not of type bool\nnon_bool_columns = X.select_dtypes(exclude='bool').columns\n\n# Scale only the non-bool columns\nX[non_bool_columns] = scaler.fit_transform(X[non_bool_columns])\ntest[non_bool_columns] = scaler.transform(test[non_bool_columns])","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras import regularizers\n\ndef rmsle(y_true, y_pred):\n    y_pred = tf.maximum(y_pred, 0)\n\n    log_true = tf.math.log1p(y_true)\n    log_pred = tf.math.log1p(y_pred)\n\n    squared_diff = tf.square(log_true - log_pred)\n    mean_squared_log_error = tf.reduce_mean(squared_diff)\n    return tf.sqrt(mean_squared_log_error)\n\n\nmodel = Sequential([\n    Dense(256, activation='relu'), \n    Dense(128, activation='relu'),\n    Dense(64, activation='relu'),\n    Dense(32, activation='relu', kernel_regularizer=regularizers.l2(0.0005)),\n    Dense(1)\n])\n\nmodel.compile(optimizer='adam', loss=rmsle)\n\nhistory = model.fit(X_train, y_train, epochs= 30, batch_size=64, validation_data=(X_val, y_val)) ","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 6))\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss Over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend(loc='upper right')\nplt.grid(True)\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## -> we will train the model with 15 epochs","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras import regularizers\n\ndef rmsle(y_true, y_pred):\n    y_pred = tf.maximum(y_pred, 0)\n\n    log_true = tf.math.log1p(y_true)\n    log_pred = tf.math.log1p(y_pred)\n\n    squared_diff = tf.square(log_true - log_pred)\n    mean_squared_log_error = tf.reduce_mean(squared_diff)\n    return tf.sqrt(mean_squared_log_error)\n\n\nmodel = Sequential([\n    Dense(256, activation='relu'), \n    Dense(128, activation='relu'),\n    Dense(64, activation='relu'),\n    Dense(32, activation='relu', kernel_regularizer=regularizers.l2(0.0005)),\n    Dense(1)\n])\n\nmodel.compile(optimizer='adam', loss=rmsle)\n\nhistory = model.fit(X_train, y_train, epochs= 15, batch_size=64, validation_data=(X_val, y_val)) ","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the test set\ntest_preds = model.predict(test)\n\nsample_submission['Premium Amount'] = test_preds\nsample_submission.to_csv('submission.csv', index=False)","metadata":{},"outputs":[],"execution_count":null}]}