{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:28:57.123633Z","iopub.execute_input":"2024-12-28T04:28:57.124409Z","iopub.status.idle":"2024-12-28T04:28:57.134471Z","shell.execute_reply.started":"2024-12-28T04:28:57.12437Z","shell.execute_reply":"2024-12-28T04:28:57.133195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:28:57.401809Z","iopub.execute_input":"2024-12-28T04:28:57.402256Z","iopub.status.idle":"2024-12-28T04:29:09.211948Z","shell.execute_reply.started":"2024-12-28T04:28:57.40222Z","shell.execute_reply":"2024-12-28T04:29:09.210801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.pipeline import Pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:28:45.275354Z","iopub.execute_input":"2024-12-28T04:28:45.275997Z","iopub.status.idle":"2024-12-28T04:28:50.337896Z","shell.execute_reply.started":"2024-12-28T04:28:45.275954Z","shell.execute_reply":"2024-12-28T04:28:50.336796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(train_df.shape)\nprint(train_df.dtypes)\n\nprint(train_df.describe())\n\nprint(train_df.describe(include=['object']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T01:50:23.371149Z","iopub.execute_input":"2024-12-28T01:50:23.371728Z","iopub.status.idle":"2024-12-28T01:50:26.231124Z","shell.execute_reply.started":"2024-12-28T01:50:23.371692Z","shell.execute_reply":"2024-12-28T01:50:26.229918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_df.hist(figsize=(15, 12), bins=30)\nplt.show()\n\nplt.figure(figsize=(12, 8))\nsns.boxplot(data=train_df.select_dtypes(include=['float64', 'int64']))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T19:46:50.946725Z","iopub.execute_input":"2024-12-27T19:46:50.947045Z","iopub.status.idle":"2024-12-27T19:46:54.23488Z","shell.execute_reply.started":"2024-12-27T19:46:50.947019Z","shell.execute_reply":"2024-12-27T19:46:54.233706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = train_df.select_dtypes(include=['object']).columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T01:57:16.495217Z","iopub.execute_input":"2024-12-28T01:57:16.495691Z","iopub.status.idle":"2024-12-28T01:57:16.665752Z","shell.execute_reply.started":"2024-12-28T01:57:16.495659Z","shell.execute_reply":"2024-12-28T01:57:16.664315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in categorical_columns:\n    print(f\"{col}:\")\n    print(train_df[col].value_counts())\n    print()\n\nfor col in categorical_columns:\n    plt.figure(figsize=(8, 6))\n    sns.countplot(x=col, data=train_df)\n    plt.title(f'{col} Distribution')\n    plt.xticks(rotation=45)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T19:46:54.237089Z","iopub.execute_input":"2024-12-27T19:46:54.237417Z","execution_failed":"2024-12-27T20:26:49.44Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(8, 6))\n# sns.histplot(train_df['Premium Amount'], kde=True, bins=30)\n# plt.title(\"Premium Amount Distribution\")\n# plt.show()\n\n# plt.figure(figsize=(8, 6))\n# sns.boxplot(x=train_df['Premium Amount'])\n# plt.title(\"Premium Amount Boxplot\")\n# plt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-27T20:26:49.441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Q1 = train_df['Premium Amount'].quantile(0.25)\nQ3 = train_df['Premium Amount'].quantile(0.75)\nIQR = Q3 - Q1\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\noutliers = train_df[(train_df['Premium Amount'] < lower_bound) | (train_df['Premium Amount'] > upper_bound)]\n\nprint(\"Outliers detected in Premium Amount:\")\nprint(outliers)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T01:57:21.137048Z","iopub.execute_input":"2024-12-28T01:57:21.137525Z","iopub.status.idle":"2024-12-28T01:57:21.244416Z","shell.execute_reply.started":"2024-12-28T01:57:21.137485Z","shell.execute_reply":"2024-12-28T01:57:21.243265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_columns = train_df.columns[train_df.isna().any()]\nnum_columns = len(nan_columns)\n\n\nn_rows = (num_columns // 3) + (1 if num_columns % 3 != 0 else 0)\nn_cols = min(3, num_columns)\n\nplt.figure(figsize=(15, n_rows * 5))\n\nfor idx, column in enumerate(nan_columns, 1):\n    plt.subplot(n_rows, n_cols, idx)\n    \n    non_nan = train_df[train_df[column].notna()]\n    nan = train_df[train_df[column].isna()]\n    \n    if 'Premium Amount' in train_df.columns:\n        sns.histplot(non_nan['Premium Amount'], kde=True, color='blue', label='Non-NaN', stat='density', bins=30)\n        sns.histplot(nan['Premium Amount'], kde=True, color='red', label='NaN', stat='density', bins=30)\n        plt.title(f'Premium Amount Distribution for {column}')\n        plt.legend()\n        plt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:16:10.645138Z","iopub.execute_input":"2024-12-28T03:16:10.645892Z","iopub.status.idle":"2024-12-28T03:17:20.649264Z","shell.execute_reply.started":"2024-12-28T03:16:10.645854Z","shell.execute_reply":"2024-12-28T03:17:20.647911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:03:41.25864Z","iopub.execute_input":"2024-12-28T04:03:41.25961Z","iopub.status.idle":"2024-12-28T04:03:42.116861Z","shell.execute_reply.started":"2024-12-28T04:03:41.259549Z","shell.execute_reply":"2024-12-28T04:03:42.115651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['is_Annual_Income_na'] = train_df['Annual Income'].isna()\ntrain_df['is_Credit_Score_na'] = train_df['Credit Score'].isna()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:03:43.951046Z","iopub.execute_input":"2024-12-28T04:03:43.951553Z","iopub.status.idle":"2024-12-28T04:03:43.961633Z","shell.execute_reply.started":"2024-12-28T04:03:43.951502Z","shell.execute_reply":"2024-12-28T04:03:43.960181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Annual Income'] = train_df['Annual Income'].fillna(train_df['Annual Income'].mean())\ntrain_df['Credit Score'] = train_df['Credit Score'].fillna(train_df['Credit Score'].mean())\ntrain_df['Health Score'] = train_df['Health Score'].fillna(train_df['Health Score'].mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:03:45.792818Z","iopub.execute_input":"2024-12-28T04:03:45.793268Z","iopub.status.idle":"2024-12-28T04:03:45.843982Z","shell.execute_reply.started":"2024-12-28T04:03:45.793223Z","shell.execute_reply":"2024-12-28T04:03:45.842749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                     'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\ncategorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', \n                       'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:29:28.077424Z","iopub.execute_input":"2024-12-28T04:29:28.077822Z","iopub.status.idle":"2024-12-28T04:29:28.084085Z","shell.execute_reply.started":"2024-12-28T04:29:28.077793Z","shell.execute_reply":"2024-12-28T04:29:28.082235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_transformer = OneHotEncoder(handle_unknown='ignore')\nnumerical_transformer = StandardScaler()\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_columns),\n        ('cat', categorical_transformer, categorical_columns)\n    ]\n)\n\npipeline = Pipeline(steps=[('preprocessor', preprocessor)])\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:30:32.823385Z","iopub.execute_input":"2024-12-28T04:30:32.823797Z","iopub.status.idle":"2024-12-28T04:30:34.162919Z","shell.execute_reply.started":"2024-12-28T04:30:32.823767Z","shell.execute_reply":"2024-12-28T04:30:34.161636Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1.4. Define feature matrix X and target vector y","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['Premium Amount', 'id'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:39:15.415452Z","iopub.execute_input":"2024-12-28T04:39:15.415935Z","iopub.status.idle":"2024-12-28T04:39:15.668669Z","shell.execute_reply.started":"2024-12-28T04:39:15.415901Z","shell.execute_reply":"2024-12-28T04:39:15.667501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1.5. Split the data into training and validation sets","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:39:31.251989Z","iopub.execute_input":"2024-12-28T04:39:31.252446Z","iopub.status.idle":"2024-12-28T04:39:32.428885Z","shell.execute_reply.started":"2024-12-28T04:39:31.252414Z","shell.execute_reply":"2024-12-28T04:39:32.427902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_transformer = OneHotEncoder(handle_unknown='ignore')\nnumerical_transformer = StandardScaler()\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_columns),\n        ('cat', categorical_transformer, categorical_columns)\n    ]\n)\npipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                           ('model', lgb.LGBMRegressor(objective='regression', metric='rmse', n_estimators=1000, learning_rate=0.05, num_leaves=31))])\npipeline.fit(X_train, y_train)\ny_pred = pipeline.predict(X_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:48:50.111276Z","iopub.execute_input":"2024-12-28T04:48:50.111758Z","iopub.status.idle":"2024-12-28T04:49:32.227752Z","shell.execute_reply.started":"2024-12-28T04:48:50.111724Z","shell.execute_reply":"2024-12-28T04:49:32.226578Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2.1. Initialize and train the LightGBM model","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2.2. Predict the target variable on the validation set","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-28T03:20:40.693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_log_error\nrmsle = mean_squared_log_error(y_valid, y_pred) ** 0.5\nprint(f\"RMSLE: {rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:49:32.229192Z","iopub.execute_input":"2024-12-28T04:49:32.229532Z","iopub.status.idle":"2024-12-28T04:49:32.249957Z","shell.execute_reply.started":"2024-12-28T04:49:32.229504Z","shell.execute_reply":"2024-12-28T04:49:32.248718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop(columns=['id'])\ntest_predictions = pipeline.predict(X_test)\nsubmission_df = pd.DataFrame({\n    'id': test_df['id'],\n    'Premium Amount': test_predictions\n})\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T04:52:58.309443Z","iopub.execute_input":"2024-12-28T04:52:58.30992Z","iopub.status.idle":"2024-12-28T04:53:25.641665Z","shell.execute_reply.started":"2024-12-28T04:52:58.309884Z","shell.execute_reply":"2024-12-28T04:53:25.640523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}