{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Data visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Machine learning\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.base import BaseEstimator, TransformerMixin\n\n# Suppress warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:22.822385Z","iopub.execute_input":"2024-12-01T15:02:22.822849Z","iopub.status.idle":"2024-12-01T15:02:22.829719Z","shell.execute_reply.started":"2024-12-01T15:02:22.8228Z","shell.execute_reply":"2024-12-01T15:02:22.828198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv').drop(['id', 'Policy Start Date'], axis = 1)\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv').drop(['id', 'Policy Start Date'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:22.8318Z","iopub.execute_input":"2024-12-01T15:02:22.832201Z","iopub.status.idle":"2024-12-01T15:02:31.055943Z","shell.execute_reply.started":"2024-12-01T15:02:22.832164Z","shell.execute_reply":"2024-12-01T15:02:31.05476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:31.057482Z","iopub.execute_input":"2024-12-01T15:02:31.057823Z","iopub.status.idle":"2024-12-01T15:02:31.085347Z","shell.execute_reply.started":"2024-12-01T15:02:31.057789Z","shell.execute_reply":"2024-12-01T15:02:31.083939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\ng = sns.displot(train_df['Premium Amount'], bins=10, kde=True)\ng.fig.suptitle('Distribution of Premium Amount', fontsize=16)\ng.set_axis_labels('Premium Amount', 'Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:31.086968Z","iopub.execute_input":"2024-12-01T15:02:31.087514Z","iopub.status.idle":"2024-12-01T15:02:36.905864Z","shell.execute_reply.started":"2024-12-01T15:02:31.087463Z","shell.execute_reply":"2024-12-01T15:02:36.904731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\ng = sns.displot(train_df['Age'], bins=range(int(train_df['Age'].min()), int(train_df['Age'].max()), 5), kde=True)\ng.fig.suptitle('Distribution of Age', fontsize=16)\ng.set_axis_labels('Age', 'Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:36.908397Z","iopub.execute_input":"2024-12-01T15:02:36.908752Z","iopub.status.idle":"2024-12-01T15:02:42.703476Z","shell.execute_reply.started":"2024-12-01T15:02:36.90872Z","shell.execute_reply":"2024-12-01T15:02:42.702425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\ng = sns.displot(train_df['Annual Income'], bins=10, kde=True)\ng.fig.suptitle('Distribution of Income', fontsize=16)\ng.set_axis_labels('Age', 'Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:42.705106Z","iopub.execute_input":"2024-12-01T15:02:42.705551Z","iopub.status.idle":"2024-12-01T15:02:48.474116Z","shell.execute_reply.started":"2024-12-01T15:02:42.705504Z","shell.execute_reply":"2024-12-01T15:02:48.473035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=train_df, x='Marital Status', palette='Set2')\n\nplt.title('Marital Status', fontsize=16)\nplt.ylabel('Count', fontsize=14)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:48.476139Z","iopub.execute_input":"2024-12-01T15:02:48.476588Z","iopub.status.idle":"2024-12-01T15:02:49.417735Z","shell.execute_reply.started":"2024-12-01T15:02:48.476539Z","shell.execute_reply":"2024-12-01T15:02:49.416672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=train_df, x='Number of Dependents', palette='Set2')\nplt.title('Count of Number of Dependents', fontsize=16)\nplt.ylabel('Count', fontsize=14)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.419566Z","iopub.execute_input":"2024-12-01T15:02:49.419994Z","iopub.status.idle":"2024-12-01T15:02:49.77072Z","shell.execute_reply.started":"2024-12-01T15:02:49.419947Z","shell.execute_reply":"2024-12-01T15:02:49.76972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DateFeatureExtractor(BaseEstimator, TransformerMixin):\n    \"\"\"Custom transformer to extract sine and cosine features from date.\"\"\"\n    \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X):\n        # Ensure 'Policy Start Date' is in datetime format\n        X['Policy Start Date'] = pd.to_datetime(X['Policy Start Date'])\n        \n        # Extract day and month\n        X['Policy Start Day'] = X['Policy Start Date'].dt.day\n        X['Policy Start Month'] = X['Policy Start Date'].dt.month\n        \n        # Create sine and cosine features\n        X['Policy Start Day_Sin'] = np.sin(2 * np.pi * X['Policy Start Day'] / 31)\n        X['Policy Start Day_Cos'] = np.cos(2 * np.pi * X['Policy Start Day'] / 31)\n        X['Policy Start Month_Sin'] = np.sin(2 * np.pi * X['Policy Start Month'] / 12)\n        X['Policy Start Month_Cos'] = np.cos(2 * np.pi * X['Policy Start Month'] / 12)\n        \n        # Drop the original date column\n        return X.drop(columns=['Policy Start Date'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.772613Z","iopub.execute_input":"2024-12-01T15:02:49.773089Z","iopub.status.idle":"2024-12-01T15:02:49.783302Z","shell.execute_reply.started":"2024-12-01T15:02:49.773039Z","shell.execute_reply":"2024-12-01T15:02:49.782088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.select_dtypes('object').columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.784945Z","iopub.execute_input":"2024-12-01T15:02:49.785296Z","iopub.status.idle":"2024-12-01T15:02:49.94013Z","shell.execute_reply.started":"2024-12-01T15:02:49.785263Z","shell.execute_reply":"2024-12-01T15:02:49.938717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_pipeline():\n    categorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n                        'Policy Type', 'Customer Feedback','Smoking Status',\n                        'Exercise Frequency', 'Property Type']\n    \n    # Create a pipeline for preprocessing\n    pipeline = Pipeline(steps=[\n        # ('date_features', DateFeatureExtractor()),  # Extract date features\n        ('encoder', ColumnTransformer(transformers=[\n            ('cat', OrdinalEncoder(), categorical_cols)  # Ordinally encode categorical features\n        ], remainder='passthrough')), # Leave other columns unchanged\n        # ('imputer', KNNImputer(n_neighbors=5)),  # Impute missing values\n    ])\n    \n    return pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.942304Z","iopub.execute_input":"2024-12-01T15:02:49.942851Z","iopub.status.idle":"2024-12-01T15:02:49.950459Z","shell.execute_reply.started":"2024-12-01T15:02:49.942803Z","shell.execute_reply":"2024-12-01T15:02:49.948779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_data(train_df, test_df):\n    pipeline = create_pipeline()\n    train_processed = pipeline.fit_transform(train_df)\n    test_processed = pipeline.transform(test_df)\n    \n    return train_processed, test_processed\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.951768Z","iopub.execute_input":"2024-12-01T15:02:49.952123Z","iopub.status.idle":"2024-12-01T15:02:49.971951Z","shell.execute_reply.started":"2024-12-01T15:02:49.952088Z","shell.execute_reply":"2024-12-01T15:02:49.970732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = train_df.pop('Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.973807Z","iopub.execute_input":"2024-12-01T15:02:49.97431Z","iopub.status.idle":"2024-12-01T15:02:49.993088Z","shell.execute_reply.started":"2024-12-01T15:02:49.974262Z","shell.execute_reply":"2024-12-01T15:02:49.991772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_processed, test_processed = preprocess_data(train_df, test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:49.996813Z","iopub.execute_input":"2024-12-01T15:02:49.997238Z","iopub.status.idle":"2024-12-01T15:02:55.10972Z","shell.execute_reply.started":"2024-12-01T15:02:49.997203Z","shell.execute_reply":"2024-12-01T15:02:55.108222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\nmodel = XGBRegressor().fit(train_processed, target.values)\npredictions = model.predict(test_processed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:02:55.111456Z","iopub.execute_input":"2024-12-01T15:02:55.111807Z","iopub.status.idle":"2024-12-01T15:03:04.671379Z","shell.execute_reply.started":"2024-12-01T15:02:55.111773Z","shell.execute_reply":"2024-12-01T15:03:04.670489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = predictions\nsubmission.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T15:03:04.672721Z","iopub.execute_input":"2024-12-01T15:03:04.673818Z","iopub.status.idle":"2024-12-01T15:03:06.074857Z","shell.execute_reply.started":"2024-12-01T15:03:04.67377Z","shell.execute_reply":"2024-12-01T15:03:06.073722Z"}},"outputs":[],"execution_count":null}]}