{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:01.595561Z","iopub.execute_input":"2024-12-31T11:43:01.596403Z","iopub.status.idle":"2024-12-31T11:43:01.603604Z","shell.execute_reply.started":"2024-12-31T11:43:01.596335Z","shell.execute_reply":"2024-12-31T11:43:01.602592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:01.606934Z","iopub.execute_input":"2024-12-31T11:43:01.607618Z","iopub.status.idle":"2024-12-31T11:43:06.935344Z","shell.execute_reply.started":"2024-12-31T11:43:01.60758Z","shell.execute_reply":"2024-12-31T11:43:06.9344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:06.937017Z","iopub.execute_input":"2024-12-31T11:43:06.937688Z","iopub.status.idle":"2024-12-31T11:43:07.415868Z","shell.execute_reply.started":"2024-12-31T11:43:06.937647Z","shell.execute_reply":"2024-12-31T11:43:07.41492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew\nnumeric_features = train_data.select_dtypes(include=['number'])\n\nskewness = numeric_features.skew()\nprint(\"Skewness of numeric features:\\n\", skewness)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:07.416902Z","iopub.execute_input":"2024-12-31T11:43:07.417183Z","iopub.status.idle":"2024-12-31T11:43:07.559302Z","shell.execute_reply.started":"2024-12-31T11:43:07.417156Z","shell.execute_reply":"2024-12-31T11:43:07.558376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s=[]\nfor col in train_data.columns:\n    if train_data[col].dtypes == 'object':\n        s.append(col)\n\nprint(s)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:07.560911Z","iopub.execute_input":"2024-12-31T11:43:07.561209Z","iopub.status.idle":"2024-12-31T11:43:07.566971Z","shell.execute_reply.started":"2024-12-31T11:43:07.56118Z","shell.execute_reply":"2024-12-31T11:43:07.566046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:07.568125Z","iopub.execute_input":"2024-12-31T11:43:07.568941Z","iopub.status.idle":"2024-12-31T11:43:07.577923Z","shell.execute_reply.started":"2024-12-31T11:43:07.568911Z","shell.execute_reply":"2024-12-31T11:43:07.577212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in s:\n    print(train_data[i].unique)\n\nprint(train_data.isnull().any())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:07.578941Z","iopub.execute_input":"2024-12-31T11:43:07.579216Z","iopub.status.idle":"2024-12-31T11:43:08.160131Z","shell.execute_reply.started":"2024-12-31T11:43:07.579191Z","shell.execute_reply":"2024-12-31T11:43:08.159105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"reference_date = pd.Timestamp('1970-01-01')\ntrain_data['Policy Start Date'] = pd.to_datetime(train_data['Policy Start Date'])\ntrain_data['Policy Start Date Numeric'] = (train_data['Policy Start Date'] - reference_date).dt.days\n\ntrain_data['Policy Start Date Numeric']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:08.161353Z","iopub.execute_input":"2024-12-31T11:43:08.161934Z","iopub.status.idle":"2024-12-31T11:43:08.561107Z","shell.execute_reply.started":"2024-12-31T11:43:08.161888Z","shell.execute_reply":"2024-12-31T11:43:08.560054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\n\ncolm = ['Gender','Marital Status','Smoking Status','Education Level','Location','Customer Feedback','Exercise Frequency']\nfor i in colm:\n    train_data[i] = le.fit_transform(train_data[i])\n\ntrain_data = pd.get_dummies(train_data,columns=['Occupation','Policy Type','Property Type'])\ntrain_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:08.562195Z","iopub.execute_input":"2024-12-31T11:43:08.562589Z","iopub.status.idle":"2024-12-31T11:43:10.49813Z","shell.execute_reply.started":"2024-12-31T11:43:08.562549Z","shell.execute_reply":"2024-12-31T11:43:10.49716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in train_data.columns:\n    train_data[col] = train_data[col].fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:10.499312Z","iopub.execute_input":"2024-12-31T11:43:10.499637Z","iopub.status.idle":"2024-12-31T11:43:10.600685Z","shell.execute_reply.started":"2024-12-31T11:43:10.499609Z","shell.execute_reply":"2024-12-31T11:43:10.599761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data.isnull().any())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:10.604351Z","iopub.execute_input":"2024-12-31T11:43:10.60472Z","iopub.status.idle":"2024-12-31T11:43:10.625611Z","shell.execute_reply.started":"2024-12-31T11:43:10.604692Z","shell.execute_reply":"2024-12-31T11:43:10.624871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"reference_date = pd.Timestamp('1970-01-01')\ntest_data['Policy Start Date'] = pd.to_datetime(test_data['Policy Start Date'])\ntest_data['Policy Start Date Numeric'] = (test_data['Policy Start Date'] - reference_date).dt.days\n\ntest_data['Policy Start Date Numeric']\n\nle = LabelEncoder()\ncolm = ['Gender', 'Marital Status', 'Smoking Status', 'Education Level', \n        'Location', 'Customer Feedback', 'Exercise Frequency']\n\nfor i in colm:\n    le.fit(test_data[i])\n    test_data[i] = le.transform(test_data[i])\n\ntest_data = pd.get_dummies(test_data, columns=['Occupation', 'Policy Type', 'Property Type'])\n\nfor col in test_data.columns:\n    test_data[col] = test_data[col].fillna(-1)\n\ntest_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:10.626757Z","iopub.execute_input":"2024-12-31T11:43:10.627009Z","iopub.status.idle":"2024-12-31T11:43:12.337118Z","shell.execute_reply.started":"2024-12-31T11:43:10.626984Z","shell.execute_reply":"2024-12-31T11:43:12.33619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"psd = train_data['Policy Start Date']\npsd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:12.338229Z","iopub.execute_input":"2024-12-31T11:43:12.338513Z","iopub.status.idle":"2024-12-31T11:43:12.345685Z","shell.execute_reply.started":"2024-12-31T11:43:12.338486Z","shell.execute_reply":"2024-12-31T11:43:12.344668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = test_data.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:12.346766Z","iopub.execute_input":"2024-12-31T11:43:12.347038Z","iopub.status.idle":"2024-12-31T11:43:12.410863Z","shell.execute_reply.started":"2024-12-31T11:43:12.347012Z","shell.execute_reply":"2024-12-31T11:43:12.410256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:12.411681Z","iopub.execute_input":"2024-12-31T11:43:12.411884Z","iopub.status.idle":"2024-12-31T11:43:12.545968Z","shell.execute_reply.started":"2024-12-31T11:43:12.411863Z","shell.execute_reply":"2024-12-31T11:43:12.545289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tt = train_data.copy()\nprint(tt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:12.547018Z","iopub.execute_input":"2024-12-31T11:43:12.547349Z","iopub.status.idle":"2024-12-31T11:43:12.840129Z","shell.execute_reply.started":"2024-12-31T11:43:12.547313Z","shell.execute_reply":"2024-12-31T11:43:12.839192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Year'] = psd.dt.year\ntrain_data['Month'] = psd.dt.month\ntrain_data['Day'] = psd.dt.day\n\ntrain_data['Month_sin'] = np.sin(2 * np.pi * train_data['Month'] / 12)\ntrain_data['Month_cos'] = np.cos(2 * np.pi * train_data['Month'] / 12)\ntrain_data['Day_sin'] = np.sin(2 * np.pi * train_data['Day'] / 31)\ntrain_data['Day_cos'] = np.cos(2 * np.pi * train_data['Day'] / 31)\n# Display the updated DataFrame\nprint(train_data.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:12.841256Z","iopub.execute_input":"2024-12-31T11:43:12.841544Z","iopub.status.idle":"2024-12-31T11:43:13.114363Z","shell.execute_reply.started":"2024-12-31T11:43:12.841517Z","shell.execute_reply":"2024-12-31T11:43:13.113453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nimport xgboost as xgb\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import KFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:13.115297Z","iopub.execute_input":"2024-12-31T11:43:13.115529Z","iopub.status.idle":"2024-12-31T11:43:13.119496Z","shell.execute_reply.started":"2024-12-31T11:43:13.115506Z","shell.execute_reply":"2024-12-31T11:43:13.118634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:13.120486Z","iopub.execute_input":"2024-12-31T11:43:13.120752Z","iopub.status.idle":"2024-12-31T11:43:13.181203Z","shell.execute_reply.started":"2024-12-31T11:43:13.120728Z","shell.execute_reply":"2024-12-31T11:43:13.180359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import catboost\nfrom catboost import CatBoostRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\n\ndef rmsle(y_true, y_pred):\n    return np.sqrt(np.mean(np.square(np.log1p(y_pred) - np.log1p(y_true))))\n\nX = train_data.drop(columns=['Premium Amount'])\ny = train_data['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:13.182192Z","iopub.execute_input":"2024-12-31T11:43:13.182501Z","iopub.status.idle":"2024-12-31T11:43:13.316487Z","shell.execute_reply.started":"2024-12-31T11:43:13.182473Z","shell.execute_reply":"2024-12-31T11:43:13.315541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.drop(columns=['Year'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:13.317818Z","iopub.execute_input":"2024-12-31T11:43:13.318433Z","iopub.status.idle":"2024-12-31T11:43:13.426528Z","shell.execute_reply.started":"2024-12-31T11:43:13.318369Z","shell.execute_reply":"2024-12-31T11:43:13.425878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, PolynomialFeatures, MinMaxScaler\n\n\npoly = PolynomialFeatures(degree=2, include_bias=False)\n\n# Step 1: Fit and transform the polynomial features\ninteraction_terms = poly.fit_transform(X[['Annual Income', 'Health Score', 'Credit Score']])\n\n# Step 2: Get the correct feature names from the fitted model\ninteraction_feature_names = poly.get_feature_names_out()\n\n# Step 3: Create a DataFrame for the interaction terms\ninteraction_df = pd.DataFrame(interaction_terms, columns=interaction_feature_names)\n\n# Step 4: Combine the new features with the original DataFrame X\nX = pd.concat([X, interaction_df], axis=1)\n\nX['Log_Vehicle_Age_by_Claims'] = np.log1p(X['Vehicle Age']) / (\n    X['Previous Claims'] + 1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:13.427627Z","iopub.execute_input":"2024-12-31T11:43:13.427969Z","iopub.status.idle":"2024-12-31T11:43:13.849095Z","shell.execute_reply.started":"2024-12-31T11:43:13.427931Z","shell.execute_reply":"2024-12-31T11:43:13.848157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"poly = PolynomialFeatures(degree=2, include_bias=False)\n\n# Fit and transform on test_data\ninteraction_terms_test = poly.fit_transform(test_data[['Annual Income', 'Health Score', 'Credit Score']])\ninteraction_feature_names_test = poly.get_feature_names_out()\ninteraction_df_test = pd.DataFrame(interaction_terms_test, columns=interaction_feature_names_test)\ntest_data = pd.concat([test_data, interaction_df_test], axis=1)\n\n# Add Log-Scaled Vehicle Age per Previous Claims in test_data\ntest_data['Log_Vehicle_Age_by_Claims'] = np.log1p(test_data['Vehicle Age']) / (test_data['Previous Claims'] + 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:13.850111Z","iopub.execute_input":"2024-12-31T11:43:13.85037Z","iopub.status.idle":"2024-12-31T11:43:14.046134Z","shell.execute_reply.started":"2024-12-31T11:43:13.850345Z","shell.execute_reply":"2024-12-31T11:43:14.045488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = test_data.loc[:, ~test_data.columns.duplicated()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:14.047292Z","iopub.execute_input":"2024-12-31T11:43:14.047692Z","iopub.status.idle":"2024-12-31T11:43:14.086375Z","shell.execute_reply.started":"2024-12-31T11:43:14.047652Z","shell.execute_reply":"2024-12-31T11:43:14.085812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.loc[:, ~X.columns.duplicated()]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:14.087351Z","iopub.execute_input":"2024-12-31T11:43:14.087629Z","iopub.status.idle":"2024-12-31T11:43:14.166654Z","shell.execute_reply.started":"2024-12-31T11:43:14.087603Z","shell.execute_reply":"2024-12-31T11:43:14.165928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:14.167559Z","iopub.execute_input":"2024-12-31T11:43:14.167822Z","iopub.status.idle":"2024-12-31T11:43:14.230977Z","shell.execute_reply.started":"2024-12-31T11:43:14.167797Z","shell.execute_reply":"2024-12-31T11:43:14.230111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_c = X.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:14.23209Z","iopub.execute_input":"2024-12-31T11:43:14.232407Z","iopub.status.idle":"2024-12-31T11:43:14.452976Z","shell.execute_reply.started":"2024-12-31T11:43:14.232358Z","shell.execute_reply":"2024-12-31T11:43:14.451915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.loc[:, 'C_H'] = X['Credit Score'] / X['Health Score']\nX.loc[:, 'H_P'] = X['Health Score'] / X['Previous Claims']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:14.454214Z","iopub.execute_input":"2024-12-31T11:43:14.454521Z","iopub.status.idle":"2024-12-31T11:43:14.468606Z","shell.execute_reply.started":"2024-12-31T11:43:14.454493Z","shell.execute_reply":"2024-12-31T11:43:14.467742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = X.corr()\n\nupper_triangle = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\nhighly_correlated = [column for column in upper_triangle.columns if any(upper_triangle[column] > 0.85)]\n\ndf_cleaned = X.drop(columns=highly_correlated)\n\n# Print the removed features\nprint(\"Removed highly correlated features:\", highly_correlated)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:14.473035Z","iopub.execute_input":"2024-12-31T11:43:14.473489Z","iopub.status.idle":"2024-12-31T11:43:19.514476Z","shell.execute_reply.started":"2024-12-31T11:43:14.473462Z","shell.execute_reply":"2024-12-31T11:43:19.513563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_remove = ['Annual Income^2', 'Annual Income Credit Score', 'Health Score^2', 'Health Score Credit Score', 'Credit Score^2']\n\nX = X.drop(columns=columns_to_remove)\nX.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:19.515666Z","iopub.execute_input":"2024-12-31T11:43:19.515939Z","iopub.status.idle":"2024-12-31T11:43:19.674166Z","shell.execute_reply.started":"2024-12-31T11:43:19.515912Z","shell.execute_reply":"2024-12-31T11:43:19.673056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_remove = ['Annual Income^2', 'Annual Income Credit Score', 'Health Score^2', 'Health Score Credit Score', 'Credit Score^2']\n\ntest_data = test_data.drop(columns=columns_to_remove)\ntest_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:19.675632Z","iopub.execute_input":"2024-12-31T11:43:19.67589Z","iopub.status.idle":"2024-12-31T11:43:19.737091Z","shell.execute_reply.started":"2024-12-31T11:43:19.675864Z","shell.execute_reply":"2024-12-31T11:43:19.736326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data['Year'] = psd.dt.year\ntest_data['Month'] = psd.dt.month\ntest_data['Day'] = psd.dt.day\n\ntest_data['Month_sin'] = np.sin(2 * np.pi * test_data['Month'] / 12)\ntest_data['Month_cos'] = np.cos(2 * np.pi * test_data['Month'] / 12)\ntest_data['Day_sin'] = np.sin(2 * np.pi * test_data['Day'] / 31)\ntest_data['Day_cos'] = np.cos(2 * np.pi * test_data['Day'] / 31)\n# Display the updated DataFrame\nprint(test_data.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:19.7381Z","iopub.execute_input":"2024-12-31T11:43:19.738353Z","iopub.status.idle":"2024-12-31T11:43:20.022234Z","shell.execute_reply.started":"2024-12-31T11:43:19.738327Z","shell.execute_reply":"2024-12-31T11:43:20.021479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = test_data.drop(columns=['Year'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.023482Z","iopub.execute_input":"2024-12-31T11:43:20.023871Z","iopub.status.idle":"2024-12-31T11:43:20.069897Z","shell.execute_reply.started":"2024-12-31T11:43:20.023826Z","shell.execute_reply":"2024-12-31T11:43:20.069274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.070957Z","iopub.execute_input":"2024-12-31T11:43:20.071291Z","iopub.status.idle":"2024-12-31T11:43:20.426863Z","shell.execute_reply.started":"2024-12-31T11:43:20.071254Z","shell.execute_reply":"2024-12-31T11:43:20.42587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train.loc[:, ~X_train.columns.duplicated()]\nX_test = X_test.loc[:, ~X_test.columns.duplicated()]\nX_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.428144Z","iopub.execute_input":"2024-12-31T11:43:20.428979Z","iopub.status.idle":"2024-12-31T11:43:20.566342Z","shell.execute_reply.started":"2024-12-31T11:43:20.428935Z","shell.execute_reply":"2024-12-31T11:43:20.565324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train_log = np.log1p(y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.567623Z","iopub.execute_input":"2024-12-31T11:43:20.567919Z","iopub.status.idle":"2024-12-31T11:43:20.573767Z","shell.execute_reply.started":"2024-12-31T11:43:20.567891Z","shell.execute_reply":"2024-12-31T11:43:20.572999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.574958Z","iopub.execute_input":"2024-12-31T11:43:20.575307Z","iopub.status.idle":"2024-12-31T11:43:20.650355Z","shell.execute_reply.started":"2024-12-31T11:43:20.575269Z","shell.execute_reply":"2024-12-31T11:43:20.649693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train['PC'] = X_train['Previous Claims']/X_train['Customer Feedback']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.651334Z","iopub.execute_input":"2024-12-31T11:43:20.651629Z","iopub.status.idle":"2024-12-31T11:43:20.659769Z","shell.execute_reply.started":"2024-12-31T11:43:20.651603Z","shell.execute_reply":"2024-12-31T11:43:20.659073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:20.660803Z","iopub.execute_input":"2024-12-31T11:43:20.661057Z","iopub.status.idle":"2024-12-31T11:43:23.745237Z","shell.execute_reply.started":"2024-12-31T11:43:20.661027Z","shell.execute_reply":"2024-12-31T11:43:23.744461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:23.746878Z","iopub.execute_input":"2024-12-31T11:43:23.747283Z","iopub.status.idle":"2024-12-31T11:43:24.083264Z","shell.execute_reply.started":"2024-12-31T11:43:23.747241Z","shell.execute_reply":"2024-12-31T11:43:24.082342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Checking X_train for NaN or infinite values...\")\nprint(pd.DataFrame(X_train).isnull().sum())  # Check for NaN\nprint(np.any(np.isinf(X_train)))  # Check for infinity\n\nprint(\"Checking y_train_log for NaN or infinite values...\")\nprint(np.any(np.isnan(y_train_log)))  # Check for NaN\nprint(np.any(np.isinf(y_train_log))) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:24.084276Z","iopub.execute_input":"2024-12-31T11:43:24.084575Z","iopub.status.idle":"2024-12-31T11:43:24.141939Z","shell.execute_reply.started":"2024-12-31T11:43:24.08454Z","shell.execute_reply":"2024-12-31T11:43:24.141037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train['PC'] = X_train['PC'].fillna(X_train['PC'].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:24.143115Z","iopub.execute_input":"2024-12-31T11:43:24.143428Z","iopub.status.idle":"2024-12-31T11:43:24.172731Z","shell.execute_reply.started":"2024-12-31T11:43:24.143381Z","shell.execute_reply":"2024-12-31T11:43:24.171789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# Replace NaN (formerly inf) with the median of each column\nX_train.fillna(X_train.median(), inplace=True)\n\n# Verify there are no inf or NaN values left\nprint(np.any(np.isinf(X_train)))  # Should now be False\nprint(X_train.isnull().sum()) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:24.173837Z","iopub.execute_input":"2024-12-31T11:43:24.174103Z","iopub.status.idle":"2024-12-31T11:43:24.993846Z","shell.execute_reply.started":"2024-12-31T11:43:24.174077Z","shell.execute_reply":"2024-12-31T11:43:24.992986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nimport time\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nimport numpy as np\nfrom optuna.samplers import CmaEsSampler\n\ndef objective(trial):\n    # Used hyperparameters\n    params = {\n        'learning_rate': trial.suggest_float('learning_rate', 0.003, 0.04, log=True),\n        'max_depth': trial.suggest_int('max_depth', 11, 22),\n        'min_child_weight': trial.suggest_int('min_child_weight', 10, 80),\n        'n_estimators': trial.suggest_int('n_estimators', 300, 1100),\n        'reg_lambda': trial.suggest_float('reg_lambda', 0.1, 10, log=True),\n        'max_delta_step': trial.suggest_int('max_delta_step', 0, 10),\n        'colsample_bylevel': trial.suggest_float('colsample_bylevel', 0.4, 0.8),\n        'max_leaves': trial.suggest_int('max_leaves', 0, 1024),\n        'tree_method': 'hist',\n        'gpu_id': 0,  # Using GPU\n        'random_state': 42 \n    }\n\n    model = XGBRegressor(**params)\n\n    cv = KFold(n_splits=5, shuffle=True, random_state=42)\n    scores = cross_val_score(model, X_train, y_train_log, cv=cv, scoring=\"neg_root_mean_squared_error\")\n\n    # Returning the average negative RMSE as the objective value\n    return -scores.mean()\n\n# Create a study\nstudy = optuna.create_study(direction=\"minimize\", sampler=optuna.samplers.TPESampler(seed=42))\n\n# Start the optimization\nstart_time = time.time()\nstudy.optimize(objective, n_trials=19)\nend_time = time.time()\n\n# Best results\nprint(\"Best params: \", study.best_params)\nprint(\"Best RMSLE: \", study.best_value)\nprint(f\"Search time: {end_time - start_time:.2f} seconds\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:43:24.995143Z","iopub.execute_input":"2024-12-31T11:43:24.995739Z","iopub.status.idle":"2024-12-31T12:18:44.475067Z","shell.execute_reply.started":"2024-12-31T11:43:24.995695Z","shell.execute_reply":"2024-12-31T12:18:44.473952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_trial = study.best_trial\nif \"best_model\" in best_trial.user_attrs:\n    best_model = best_trial.user_attrs[\"best_model\"]\nelse:\n    print(\"The best model was not stored in the trial's user attributes.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:18:44.476282Z","iopub.execute_input":"2024-12-31T12:18:44.47659Z","iopub.status.idle":"2024-12-31T12:18:44.482016Z","shell.execute_reply.started":"2024-12-31T12:18:44.47656Z","shell.execute_reply":"2024-12-31T12:18:44.48115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nfrom xgboost import XGBRegressor\n\n# Train the best model with the optimal parameters\nbest_params = study.best_params\nbest_model = XGBRegressor(**best_params)\nbest_model.fit(X_train, y_train_log)\n\n# Get feature importances\nfeature_importances = best_model.feature_importances_\n\n# Create a DataFrame for better handling\nimportance_df = pd.DataFrame({\n    'Feature': X_train.columns,\n    'Importance': feature_importances\n})\n\n# Sort by importance and take the top 20\ntop_features = importance_df.sort_values(by='Importance', ascending=True).head(20)\n\n# Plot the top 20 features\nplt.figure(figsize=(10, 8))\nplt.barh(top_features['Feature'], top_features['Importance'], color='skyblue')\nplt.xlabel('Feature Importance')\nplt.ylabel('Features')\nplt.title('Top 20 Feature Importances')\nplt.gca().invert_yaxis()  # Invert y-axis to show the highest importance on top\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:18:44.483237Z","iopub.execute_input":"2024-12-31T12:18:44.483949Z","iopub.status.idle":"2024-12-31T12:20:04.224693Z","shell.execute_reply.started":"2024-12-31T12:18:44.483906Z","shell.execute_reply":"2024-12-31T12:20:04.223839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the last 6 features after sorting by importance\nlast_6_features = importance_df.sort_values(by='Importance', ascending=False).tail(6)['Feature'].tolist()\n\n# Print the list of last 6 features\nprint(\"Last 6 features:\", last_6_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.225737Z","iopub.execute_input":"2024-12-31T12:20:04.225998Z","iopub.status.idle":"2024-12-31T12:20:04.231765Z","shell.execute_reply.started":"2024-12-31T12:20:04.225973Z","shell.execute_reply":"2024-12-31T12:20:04.230894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data['C_H'] = test_data['Credit Score'] / test_data['Health Score']\ntest_data['H_P'] = test_data['Health Score'] / test_data['Previous Claims']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.232831Z","iopub.execute_input":"2024-12-31T12:20:04.233069Z","iopub.status.idle":"2024-12-31T12:20:04.247733Z","shell.execute_reply.started":"2024-12-31T12:20:04.233045Z","shell.execute_reply":"2024-12-31T12:20:04.246834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test_data.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.248723Z","iopub.execute_input":"2024-12-31T12:20:04.248966Z","iopub.status.idle":"2024-12-31T12:20:04.297855Z","shell.execute_reply.started":"2024-12-31T12:20:04.248943Z","shell.execute_reply":"2024-12-31T12:20:04.297055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.298781Z","iopub.execute_input":"2024-12-31T12:20:04.299024Z","iopub.status.idle":"2024-12-31T12:20:04.34158Z","shell.execute_reply.started":"2024-12-31T12:20:04.299Z","shell.execute_reply":"2024-12-31T12:20:04.340792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.342563Z","iopub.execute_input":"2024-12-31T12:20:04.342902Z","iopub.status.idle":"2024-12-31T12:20:04.389511Z","shell.execute_reply.started":"2024-12-31T12:20:04.342863Z","shell.execute_reply":"2024-12-31T12:20:04.388692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_features = set(X.columns) - set(test_data.columns)\nextra_features = set(test_data.columns) - set(X.columns)\n\nprint(\"Missing features in test_data:\", missing_features)\nprint(\"Extra features in test_data:\", extra_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.390648Z","iopub.execute_input":"2024-12-31T12:20:04.391327Z","iopub.status.idle":"2024-12-31T12:20:04.396158Z","shell.execute_reply.started":"2024-12-31T12:20:04.391284Z","shell.execute_reply":"2024-12-31T12:20:04.395285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['PC'] = test['Previous Claims']/test['Customer Feedback']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.39721Z","iopub.execute_input":"2024-12-31T12:20:04.397579Z","iopub.status.idle":"2024-12-31T12:20:04.411349Z","shell.execute_reply.started":"2024-12-31T12:20:04.397546Z","shell.execute_reply":"2024-12-31T12:20:04.41071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = X_test.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.412339Z","iopub.execute_input":"2024-12-31T12:20:04.412684Z","iopub.status.idle":"2024-12-31T12:20:04.428646Z","shell.execute_reply.started":"2024-12-31T12:20:04.412648Z","shell.execute_reply":"2024-12-31T12:20:04.428041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test['PC'] = X_test['Previous Claims']/X_test['Customer Feedback']\nX_test.isnull().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.429562Z","iopub.execute_input":"2024-12-31T12:20:04.429797Z","iopub.status.idle":"2024-12-31T12:20:04.442299Z","shell.execute_reply.started":"2024-12-31T12:20:04.429773Z","shell.execute_reply":"2024-12-31T12:20:04.441535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.replace([np.inf, -np.inf], np.nan, inplace=True)  # Replace infinities with NaN\nX.fillna(X.mean(), inplace=True)  # Replace NaN with column mean\nX.isnull().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.443321Z","iopub.execute_input":"2024-12-31T12:20:04.443598Z","iopub.status.idle":"2024-12-31T12:20:04.926328Z","shell.execute_reply.started":"2024-12-31T12:20:04.443574Z","shell.execute_reply":"2024-12-31T12:20:04.925469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.927626Z","iopub.execute_input":"2024-12-31T12:20:04.928024Z","iopub.status.idle":"2024-12-31T12:20:04.973782Z","shell.execute_reply.started":"2024-12-31T12:20:04.927982Z","shell.execute_reply":"2024-12-31T12:20:04.973039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.974639Z","iopub.execute_input":"2024-12-31T12:20:04.974862Z","iopub.status.idle":"2024-12-31T12:20:04.994766Z","shell.execute_reply.started":"2024-12-31T12:20:04.97484Z","shell.execute_reply":"2024-12-31T12:20:04.994001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Best parameters found by Optuna\n# Align the columns of X_test to match X_train\nX_test = X_test[X_train.columns]\n\nbest_params = study.best_params\n\n# Re-train the model using the best hyperparameters\nbest_model = XGBRegressor(**best_params)\nbest_model.fit(X_train, y_train_log)\n\n# Now, use the model for predictions (e.g., on the test set)\ny_pred = best_model.predict(X_test)\n\n# Evaluate the model performance (e.g., RMSE or any other metric)\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\nprint(\"RMSE on Test Set:\", rmse)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:20:04.995889Z","iopub.execute_input":"2024-12-31T12:20:04.996562Z","iopub.status.idle":"2024-12-31T12:21:27.254446Z","shell.execute_reply.started":"2024-12-31T12:20:04.99652Z","shell.execute_reply":"2024-12-31T12:21:27.253631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.isna().sum())  # Check for NaN values\nprint(np.isinf(test).sum())  # Check for infinity values\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:27.255258Z","iopub.execute_input":"2024-12-31T12:21:27.255536Z","iopub.status.idle":"2024-12-31T12:21:27.322168Z","shell.execute_reply.started":"2024-12-31T12:21:27.255507Z","shell.execute_reply":"2024-12-31T12:21:27.321358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.replace([np.inf, -np.inf], np.nan, inplace=True)  # Replace infinities with NaN\ntest.fillna(test.mean(), inplace=True)  # Replace NaN with column mean\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:27.323438Z","iopub.execute_input":"2024-12-31T12:21:27.324047Z","iopub.status.idle":"2024-12-31T12:21:27.584592Z","shell.execute_reply.started":"2024-12-31T12:21:27.324018Z","shell.execute_reply":"2024-12-31T12:21:27.583593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test[X_train.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:27.585938Z","iopub.execute_input":"2024-12-31T12:21:27.586209Z","iopub.status.idle":"2024-12-31T12:21:27.627221Z","shell.execute_reply.started":"2024-12-31T12:21:27.586183Z","shell.execute_reply":"2024-12-31T12:21:27.626374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_predd = best_model.predict(test)\nprint(y_predd)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:27.628221Z","iopub.execute_input":"2024-12-31T12:21:27.628507Z","iopub.status.idle":"2024-12-31T12:21:45.44738Z","shell.execute_reply.started":"2024-12-31T12:21:27.628482Z","shell.execute_reply":"2024-12-31T12:21:45.446672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_actual = np.expm1(y_predd)\n\n# Print the actual values\nprint(y_pred_actual)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:45.448144Z","iopub.execute_input":"2024-12-31T12:21:45.448404Z","iopub.status.idle":"2024-12-31T12:21:45.455484Z","shell.execute_reply.started":"2024-12-31T12:21:45.448362Z","shell.execute_reply":"2024-12-31T12:21:45.453629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame({\n    'id': test_data['id'],\n    'Premium Amount': y_pred_actual\n})\n\nsubmission_df.to_csv('subpppr.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T12:21:45.456451Z","iopub.execute_input":"2024-12-31T12:21:45.457573Z","iopub.status.idle":"2024-12-31T12:21:46.407244Z","shell.execute_reply.started":"2024-12-31T12:21:45.45752Z","shell.execute_reply":"2024-12-31T12:21:46.406591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}