{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:01:01.831097Z","iopub.execute_input":"2024-12-04T10:01:01.831504Z","iopub.status.idle":"2024-12-04T10:01:02.284385Z","shell.execute_reply.started":"2024-12-04T10:01:01.831465Z","shell.execute_reply":"2024-12-04T10:01:02.282307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import essential libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nimport warnings\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:01:44.866129Z","iopub.execute_input":"2024-12-04T10:01:44.867466Z","iopub.status.idle":"2024-12-04T10:01:44.874394Z","shell.execute_reply.started":"2024-12-04T10:01:44.867416Z","shell.execute_reply":"2024-12-04T10:01:44.873134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load datasets\ntrain = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n\n# Display the first few rows of the training dataset\ntrain.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:01:45.965643Z","iopub.execute_input":"2024-12-04T10:01:45.966096Z","iopub.status.idle":"2024-12-04T10:01:53.92743Z","shell.execute_reply.started":"2024-12-04T10:01:45.966026Z","shell.execute_reply":"2024-12-04T10:01:53.926253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the structure of the training data\ntrain.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:02:46.547188Z","iopub.execute_input":"2024-12-04T10:02:46.547604Z","iopub.status.idle":"2024-12-04T10:02:47.198312Z","shell.execute_reply.started":"2024-12-04T10:02:46.547566Z","shell.execute_reply":"2024-12-04T10:02:47.196846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count missing values\nmissing_data = train.isnull().sum()\nprint(missing_data)\n\n# Visualize missing values\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nsns.heatmap(train.isnull(), cbar=False, cmap=\"viridis\")\nplt.title(\"Missing Values Heatmap\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:02:10.159704Z","iopub.execute_input":"2024-12-04T10:02:10.160149Z","iopub.status.idle":"2024-12-04T10:02:34.894818Z","shell.execute_reply.started":"2024-12-04T10:02:10.160107Z","shell.execute_reply":"2024-12-04T10:02:34.893529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of the target variable\nplt.figure(figsize=(10, 6))\nsns.histplot(train['Premium Amount'], kde=True, color=\"blue\", bins=30)\nplt.title(\"Distribution of Premium Amount\")\nplt.xlabel(\"Premium Amount\")\nplt.ylabel(\"Frequency\")\nplt.show()\n\n# Check for skewness in the target\nprint(\"Skewness:\", train['Premium Amount'].skew())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:02:50.586216Z","iopub.execute_input":"2024-12-04T10:02:50.586622Z","iopub.status.idle":"2024-12-04T10:02:56.280712Z","shell.execute_reply.started":"2024-12-04T10:02:50.586583Z","shell.execute_reply":"2024-12-04T10:02:56.279514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot histograms for numerical features\nnum_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n            'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\ntrain[num_cols].hist(figsize=(15, 10), bins=20, color='skyblue', edgecolor='black')\nplt.suptitle(\"Numerical Feature Distributions\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:02:59.480615Z","iopub.execute_input":"2024-12-04T10:02:59.481128Z","iopub.status.idle":"2024-12-04T10:03:01.360058Z","shell.execute_reply.started":"2024-12-04T10:02:59.481081Z","shell.execute_reply":"2024-12-04T10:03:01.358858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation heatmap\nplt.figure(figsize=(10, 8))\nsns.heatmap(train[num_cols].corr(), annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title(\"Correlation Heatmap of Numerical Features\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:07.241977Z","iopub.execute_input":"2024-12-04T10:03:07.242425Z","iopub.status.idle":"2024-12-04T10:03:07.968252Z","shell.execute_reply.started":"2024-12-04T10:03:07.242387Z","shell.execute_reply":"2024-12-04T10:03:07.967117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n            'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', \n            'Exercise Frequency', 'Property Type']\n\n# Plot count plots for categorical features\nfor col in cat_cols:\n    plt.figure(figsize=(8, 4))\n    sns.countplot(data=train, x=col, order=train[col].value_counts().index, palette=\"viridis\")\n    plt.title(f\"Distribution of {col}\")\n    plt.xticks(rotation=45)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:13.023167Z","iopub.execute_input":"2024-12-04T10:03:13.023561Z","iopub.status.idle":"2024-12-04T10:03:20.354565Z","shell.execute_reply.started":"2024-12-04T10:03:13.023527Z","shell.execute_reply":"2024-12-04T10:03:20.353299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Boxplots for categorical features vs target\nfor col in cat_cols:\n    plt.figure(figsize=(8, 4))\n    sns.boxplot(data=train, x=col, y='Premium Amount', palette=\"viridis\")\n    plt.title(f\"Premium Amount vs {col}\")\n    plt.xticks(rotation=45)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:26.7214Z","iopub.execute_input":"2024-12-04T10:03:26.721826Z","iopub.status.idle":"2024-12-04T10:03:35.10942Z","shell.execute_reply.started":"2024-12-04T10:03:26.72178Z","shell.execute_reply":"2024-12-04T10:03:35.108135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation of numerical features with the target\ncorrelation_with_target = train[num_cols + ['Premium Amount']].corr()['Premium Amount'].sort_values(ascending=False)\nprint(correlation_with_target)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:39.588887Z","iopub.execute_input":"2024-12-04T10:03:39.589602Z","iopub.status.idle":"2024-12-04T10:03:39.938318Z","shell.execute_reply.started":"2024-12-04T10:03:39.589562Z","shell.execute_reply":"2024-12-04T10:03:39.937119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mean Premium Amount for each category\nfor col in cat_cols:\n    print(f\"Mean Premium Amount by {col}:\")\n    print(train.groupby(col)['Premium Amount'].mean().sort_values(ascending=False))\n    print()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:40.943336Z","iopub.execute_input":"2024-12-04T10:03:40.943723Z","iopub.status.idle":"2024-12-04T10:03:41.840651Z","shell.execute_reply.started":"2024-12-04T10:03:40.943686Z","shell.execute_reply":"2024-12-04T10:03:41.839484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Boxplot for each numerical column\nfor col in num_cols:\n    plt.figure(figsize=(8, 4))\n    sns.boxplot(data=train, x=col, palette=\"viridis\")\n    plt.title(f\"Boxplot of {col}\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:42.294314Z","iopub.execute_input":"2024-12-04T10:03:42.294718Z","iopub.status.idle":"2024-12-04T10:03:44.314195Z","shell.execute_reply.started":"2024-12-04T10:03:42.294682Z","shell.execute_reply":"2024-12-04T10:03:44.312911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import zscore\n\n# Calculate Z-scores for numerical columns\nz_scores = train[num_cols].apply(zscore)\n\n# Find rows where Z-score > 3\noutliers = (z_scores > 3).sum(axis=1)\nprint(f\"Number of rows with outliers: {outliers[outliers > 0].count()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:50.660479Z","iopub.execute_input":"2024-12-04T10:03:50.660877Z","iopub.status.idle":"2024-12-04T10:03:50.96329Z","shell.execute_reply.started":"2024-12-04T10:03:50.66084Z","shell.execute_reply":"2024-12-04T10:03:50.96206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove outliers based on Z-score\ntrain_no_outliers = train[(z_scores <= 3).all(axis=1)]\n\n# Compare original and cleaned data\nprint(f\"Original data shape: {train.shape}\")\nprint(f\"Data shape after removing outliers: {train_no_outliers.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:52.439754Z","iopub.execute_input":"2024-12-04T10:03:52.440328Z","iopub.status.idle":"2024-12-04T10:03:52.460387Z","shell.execute_reply.started":"2024-12-04T10:03:52.440284Z","shell.execute_reply":"2024-12-04T10:03:52.459119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mean target value for each category\nfor col in cat_cols:\n    print(f\"Mean Premium Amount by {col}:\")\n    print(train.groupby(col)['Premium Amount'].mean().sort_values(ascending=False))\n    print()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:04:21.47346Z","iopub.execute_input":"2024-12-04T10:04:21.4739Z","iopub.status.idle":"2024-12-04T10:04:22.375845Z","shell.execute_reply.started":"2024-12-04T10:04:21.473861Z","shell.execute_reply":"2024-12-04T10:04:22.374866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Boxplot with categorical features\nfor col in ['Marital Status', 'Policy Type']:\n    plt.figure(figsize=(8, 4))\n    sns.boxplot(data=train, x=col, y='Premium Amount', hue=\"Gender\", palette=\"coolwarm\")\n    plt.title(f\"Premium Amount vs {col}\")\n    plt.xticks(rotation=45)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:04:29.683448Z","iopub.execute_input":"2024-12-04T10:04:29.684372Z","iopub.status.idle":"2024-12-04T10:04:33.189287Z","shell.execute_reply.started":"2024-12-04T10:04:29.684322Z","shell.execute_reply":"2024-12-04T10:04:33.188093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categorical variables numerically\ncat_encoded = train[cat_cols].apply(lambda x: x.astype('category').cat.codes)\n\n# Correlation heatmap\nplt.figure(figsize=(10, 8))\nsns.heatmap(cat_encoded.corr(), annot=True, cmap=\"coolwarm\")\nplt.title(\"Correlation Heatmap for Categorical Features\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:04:50.32098Z","iopub.execute_input":"2024-12-04T10:04:50.3221Z","iopub.status.idle":"2024-12-04T10:04:52.206368Z","shell.execute_reply.started":"2024-12-04T10:04:50.322025Z","shell.execute_reply":"2024-12-04T10:04:52.205125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Crosstab between two categorical variables\ncross_tab = pd.crosstab(train['Policy Type'], train['Marital Status'])\nprint(cross_tab)\n\n# Heatmap for crosstab\nsns.heatmap(cross_tab, annot=True, cmap=\"coolwarm\")\nplt.title(\"Policy Type vs Marital Status\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:04:54.358024Z","iopub.execute_input":"2024-12-04T10:04:54.359483Z","iopub.status.idle":"2024-12-04T10:04:54.851278Z","shell.execute_reply.started":"2024-12-04T10:04:54.359436Z","shell.execute_reply":"2024-12-04T10:04:54.85013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols_with_missing = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                         'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n\nfor col in num_cols_with_missing:\n    train[col].fillna(train[col].median(), inplace=True)\n    test[col].fillna(test[col].median(), inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:05.444954Z","iopub.execute_input":"2024-12-04T10:05:05.446631Z","iopub.status.idle":"2024-12-04T10:05:05.872627Z","shell.execute_reply.started":"2024-12-04T10:05:05.446579Z","shell.execute_reply":"2024-12-04T10:05:05.87146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols_with_missing = ['Marital Status', 'Occupation', 'Customer Feedback']\n\nfor col in cat_cols_with_missing:\n    train[col].fillna(\"Unknown\", inplace=True)\n    test[col].fillna(\"Unknown\", inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:10.601205Z","iopub.execute_input":"2024-12-04T10:05:10.601612Z","iopub.status.idle":"2024-12-04T10:05:10.941397Z","shell.execute_reply.started":"2024-12-04T10:05:10.601576Z","shell.execute_reply":"2024-12-04T10:05:10.940335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if there are any remaining missing values\nprint(train.isnull().sum())\nprint(test.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:12.771455Z","iopub.execute_input":"2024-12-04T10:05:12.77188Z","iopub.status.idle":"2024-12-04T10:05:13.836093Z","shell.execute_reply.started":"2024-12-04T10:05:12.771837Z","shell.execute_reply":"2024-12-04T10:05:13.834771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Target variable\ntarget = \"Premium Amount\"\n\n# Features (drop target column)\nX = train.drop(columns=[target])\ny = np.log1p(train[target])  # Log-transform target for RMSLE\n\n# Identify categorical and numerical columns\nnum_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n            'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\ncat_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n            'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', \n            'Exercise Frequency', 'Property Type']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:14.774017Z","iopub.execute_input":"2024-12-04T10:05:14.774438Z","iopub.status.idle":"2024-12-04T10:05:14.967522Z","shell.execute_reply.started":"2024-12-04T10:05:14.774399Z","shell.execute_reply":"2024-12-04T10:05:14.966418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessor for numerical and categorical data\npreprocessor = ColumnTransformer([\n    (\"num\", StandardScaler(), num_cols),           # Scale numerical features\n    (\"cat\", OneHotEncoder(handle_unknown=\"ignore\"), cat_cols)  # Encode categorical features\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:16.803935Z","iopub.execute_input":"2024-12-04T10:05:16.804958Z","iopub.status.idle":"2024-12-04T10:05:16.809861Z","shell.execute_reply.started":"2024-12-04T10:05:16.804915Z","shell.execute_reply":"2024-12-04T10:05:16.808756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:22.396925Z","iopub.execute_input":"2024-12-04T10:05:22.39736Z","iopub.status.idle":"2024-12-04T10:05:22.40323Z","shell.execute_reply.started":"2024-12-04T10:05:22.397323Z","shell.execute_reply":"2024-12-04T10:05:22.401993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train-test split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:23.615146Z","iopub.execute_input":"2024-12-04T10:05:23.615531Z","iopub.status.idle":"2024-12-04T10:05:24.712531Z","shell.execute_reply.started":"2024-12-04T10:05:23.615498Z","shell.execute_reply":"2024-12-04T10:05:24.711518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a pipeline with preprocessing and the model\nmodel = Pipeline(steps=[\n    (\"preprocessor\", preprocessor),\n    (\"regressor\", GradientBoostingRegressor(random_state=42))\n])\n\n# Train the model\nmodel.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:26.029684Z","iopub.execute_input":"2024-12-04T10:05:26.0301Z","iopub.status.idle":"2024-12-04T10:13:08.928925Z","shell.execute_reply.started":"2024-12-04T10:05:26.030065Z","shell.execute_reply":"2024-12-04T10:13:08.927561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on validation data\nval_preds = model.predict(X_val)\n\n# Reverse log-transform predictions\nval_preds_exp = np.expm1(val_preds)\ny_val_exp = np.expm1(y_val)\n\n# Calculate RMSLE\nrmsle = np.sqrt(mean_squared_log_error(y_val_exp, val_preds_exp))\nprint(f\"Validation RMSLE: {rmsle}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:15:16.013063Z","iopub.execute_input":"2024-12-04T10:15:16.013523Z","iopub.status.idle":"2024-12-04T10:15:17.405144Z","shell.execute_reply.started":"2024-12-04T10:15:16.013483Z","shell.execute_reply":"2024-12-04T10:15:17.403816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test data\ntest_preds = model.predict(test)\ntest_preds_exp = np.expm1(test_preds)  # Reverse log-transform\n\n# Prepare the submission file\nsubmission = pd.DataFrame({\"id\": test[\"id\"], \"Premium Amount\": test_preds_exp})\nsubmission.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file created!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:15:18.776113Z","iopub.execute_input":"2024-12-04T10:15:18.776504Z","iopub.status.idle":"2024-12-04T10:15:25.023601Z","shell.execute_reply.started":"2024-12-04T10:15:18.776468Z","shell.execute_reply":"2024-12-04T10:15:25.022142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}