{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:08:42.793644Z","iopub.execute_input":"2024-12-01T06:08:42.795004Z","iopub.status.idle":"2024-12-01T06:08:42.804661Z","shell.execute_reply.started":"2024-12-01T06:08:42.794945Z","shell.execute_reply":"2024-12-01T06:08:42.80345Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:08:42.809747Z","iopub.execute_input":"2024-12-01T06:08:42.810224Z","iopub.status.idle":"2024-12-01T06:08:42.828214Z","shell.execute_reply.started":"2024-12-01T06:08:42.810175Z","shell.execute_reply":"2024-12-01T06:08:42.826875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading the dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:07.357112Z","iopub.execute_input":"2024-12-01T06:22:07.357593Z","iopub.status.idle":"2024-12-01T06:22:15.463928Z","shell.execute_reply.started":"2024-12-01T06:22:07.357559Z","shell.execute_reply":"2024-12-01T06:22:15.462855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:15.465708Z","iopub.execute_input":"2024-12-01T06:22:15.466095Z","iopub.status.idle":"2024-12-01T06:22:15.492401Z","shell.execute_reply.started":"2024-12-01T06:22:15.466059Z","shell.execute_reply":"2024-12-01T06:22:15.491107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:15.494764Z","iopub.execute_input":"2024-12-01T06:22:15.495156Z","iopub.status.idle":"2024-12-01T06:22:16.155366Z","shell.execute_reply.started":"2024-12-01T06:22:15.495122Z","shell.execute_reply":"2024-12-01T06:22:16.153972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.drop(columns = ['Policy Start Date', 'id'], axis = 1)\ntest_df = test_df.drop(columns = ['Policy Start Date', 'id'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:16.158228Z","iopub.execute_input":"2024-12-01T06:22:16.158695Z","iopub.status.idle":"2024-12-01T06:22:16.434315Z","shell.execute_reply.started":"2024-12-01T06:22:16.158644Z","shell.execute_reply":"2024-12-01T06:22:16.433294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:16.43578Z","iopub.execute_input":"2024-12-01T06:22:16.436263Z","iopub.status.idle":"2024-12-01T06:22:16.444129Z","shell.execute_reply.started":"2024-12-01T06:22:16.436217Z","shell.execute_reply":"2024-12-01T06:22:16.442711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:16.445872Z","iopub.execute_input":"2024-12-01T06:22:16.446467Z","iopub.status.idle":"2024-12-01T06:22:17.032581Z","shell.execute_reply.started":"2024-12-01T06:22:16.446432Z","shell.execute_reply":"2024-12-01T06:22:17.03142Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Impute missing values for numerical columns","metadata":{}},{"cell_type":"code","source":"numerical_cols = train_df.select_dtypes(include=['float64']).columns\nimputer = SimpleImputer(strategy='median')\ntrain_df[numerical_cols] = imputer.fit_transform(train_df[numerical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:17.033962Z","iopub.execute_input":"2024-12-01T06:22:17.034306Z","iopub.status.idle":"2024-12-01T06:22:19.177091Z","shell.execute_reply.started":"2024-12-01T06:22:17.034275Z","shell.execute_reply":"2024-12-01T06:22:19.175924Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute missing values for categorical columns","metadata":{}},{"cell_type":"code","source":"categorical_cols = train_df.select_dtypes(include=['object']).columns\nfor col in categorical_cols:\n    train_df[col] = train_df[col].fillna(train_df[col].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:19.179233Z","iopub.execute_input":"2024-12-01T06:22:19.179531Z","iopub.status.idle":"2024-12-01T06:22:21.012223Z","shell.execute_reply.started":"2024-12-01T06:22:19.179502Z","shell.execute_reply":"2024-12-01T06:22:21.010856Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding categorical variables","metadata":{}},{"cell_type":"code","source":"label_encoders = {}\nfor col in categorical_cols:\n    le = LabelEncoder()\n    train_df[col] = le.fit_transform(train_df[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:21.01369Z","iopub.execute_input":"2024-12-01T06:22:21.014069Z","iopub.status.idle":"2024-12-01T06:22:23.365614Z","shell.execute_reply.started":"2024-12-01T06:22:21.014029Z","shell.execute_reply":"2024-12-01T06:22:23.364578Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature scaling","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\ntrain_df[numerical_cols] = scaler.fit_transform(train_df[numerical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:08:59.571006Z","iopub.execute_input":"2024-12-01T06:08:59.571345Z","iopub.status.idle":"2024-12-01T06:08:59.878324Z","shell.execute_reply.started":"2024-12-01T06:08:59.571313Z","shell.execute_reply":"2024-12-01T06:08:59.877004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Split into features (X) and target (y)","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['Premium Amount'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:23.367405Z","iopub.execute_input":"2024-12-01T06:22:23.367769Z","iopub.status.idle":"2024-12-01T06:22:23.493464Z","shell.execute_reply.started":"2024-12-01T06:22:23.367732Z","shell.execute_reply":"2024-12-01T06:22:23.492392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:23.494829Z","iopub.execute_input":"2024-12-01T06:22:23.495195Z","iopub.status.idle":"2024-12-01T06:22:23.945477Z","shell.execute_reply.started":"2024-12-01T06:22:23.495159Z","shell.execute_reply":"2024-12-01T06:22:23.944227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Linear Regression","metadata":{}},{"cell_type":"code","source":"linear_model = LinearRegression()\nlinear_model.fit(X_train, y_train)\nlinear_pred = linear_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:23.947589Z","iopub.execute_input":"2024-12-01T06:22:23.947977Z","iopub.status.idle":"2024-12-01T06:22:24.856056Z","shell.execute_reply.started":"2024-12-01T06:22:23.947939Z","shell.execute_reply":"2024-12-01T06:22:24.854497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_rmse = mean_squared_error(y_test, linear_pred, squared=False)\nprint(f'Linear Regression RMSE: {linear_rmse}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:24.858256Z","iopub.execute_input":"2024-12-01T06:22:24.858866Z","iopub.status.idle":"2024-12-01T06:22:24.875576Z","shell.execute_reply.started":"2024-12-01T06:22:24.858799Z","shell.execute_reply":"2024-12-01T06:22:24.873304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_r2 = r2_score(y_test, linear_pred)\nprint(f'Linear Regression R2: {linear_r2}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:26.095789Z","iopub.execute_input":"2024-12-01T06:22:26.096693Z","iopub.status.idle":"2024-12-01T06:22:26.105784Z","shell.execute_reply.started":"2024-12-01T06:22:26.096642Z","shell.execute_reply":"2024-12-01T06:22:26.104585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Decision Trees","metadata":{}},{"cell_type":"code","source":"dt_model = DecisionTreeRegressor(random_state=42)\ndt_model.fit(X_train, y_train)\ndt_pred = dt_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:28.394722Z","iopub.execute_input":"2024-12-01T06:22:28.395157Z","iopub.status.idle":"2024-12-01T06:22:50.166003Z","shell.execute_reply.started":"2024-12-01T06:22:28.395114Z","shell.execute_reply":"2024-12-01T06:22:50.165086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dt_rmse = mean_squared_error(y_test, dt_pred, squared=False)\nprint(f'Decision Tree RMSE: {dt_rmse}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:50.16774Z","iopub.execute_input":"2024-12-01T06:22:50.168103Z","iopub.status.idle":"2024-12-01T06:22:50.176392Z","shell.execute_reply.started":"2024-12-01T06:22:50.168071Z","shell.execute_reply":"2024-12-01T06:22:50.1751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dt_r2 = r2_score(y_test, dt_pred)\nprint(f'Decision Tree R2: {dt_r2}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:22:50.178055Z","iopub.execute_input":"2024-12-01T06:22:50.178515Z","iopub.status.idle":"2024-12-01T06:22:50.19418Z","shell.execute_reply.started":"2024-12-01T06:22:50.178463Z","shell.execute_reply":"2024-12-01T06:22:50.192874Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Interpretation of RESULTS","metadata":{}},{"cell_type":"markdown","source":"* Linear Regression is performing very poorly in this case, with an R² value close to 0 and an RMSE that suggests significant error. This could imply that the linear relationships between the features and the Premium Amount are too weak or non-existent.\n* Decision Tree is performing even worse, as indicated by the negative R² value. This suggests that the decision tree might be overfitting the training data, or the data doesn't contain enough clear patterns for the model to learn useful relationships.","metadata":{}}]}