{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10172907,"sourceType":"datasetVersion","datasetId":6283072}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:30.533121Z","iopub.execute_input":"2024-12-31T20:04:30.533476Z","iopub.status.idle":"2024-12-31T20:04:30.947399Z","shell.execute_reply.started":"2024-12-31T20:04:30.533449Z","shell.execute_reply":"2024-12-31T20:04:30.946025Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing Modules:","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, OrdinalEncoder\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import make_scorer\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nimport optuna\nfrom sklearn.metrics import mean_squared_error, r2_score\nimport warnings \nwarnings.filterwarnings('ignore')\n# Disable LightGBM warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nimport logging\nlogging.getLogger('lightgbm').setLevel(logging.INFO)\nlogging.getLogger('lightgbm').setLevel(logging.ERROR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:30.949549Z","iopub.execute_input":"2024-12-31T20:04:30.950128Z","iopub.status.idle":"2024-12-31T20:04:32.475067Z","shell.execute_reply.started":"2024-12-31T20:04:30.950091Z","shell.execute_reply":"2024-12-31T20:04:32.473704Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing Data:","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(r\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_data = pd.read_csv(r\"/kaggle/input/playground-series-s4e12/test.csv\")\noriginal_data = pd.read_csv(r\"/kaggle/input/insurance-prediction-data/Insurance Premium Prediction Dataset.csv\")\ndata = pd.read_csv(r\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\n\nprint(\"train_data shape :\",train_data.shape)\nprint(\"test_data shape :\",test_data.shape)\nprint(\"original_data shape :\",original_data.shape)\nprint(\"data shape :\",data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:32.476496Z","iopub.execute_input":"2024-12-31T20:04:32.47701Z","iopub.status.idle":"2024-12-31T20:04:41.147566Z","shell.execute_reply.started":"2024-12-31T20:04:32.47698Z","shell.execute_reply":"2024-12-31T20:04:41.146344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:41.149718Z","iopub.execute_input":"2024-12-31T20:04:41.150051Z","iopub.status.idle":"2024-12-31T20:04:41.180195Z","shell.execute_reply.started":"2024-12-31T20:04:41.150023Z","shell.execute_reply":"2024-12-31T20:04:41.178996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.isna().sum().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:41.181348Z","iopub.execute_input":"2024-12-31T20:04:41.181645Z","iopub.status.idle":"2024-12-31T20:04:41.815461Z","shell.execute_reply.started":"2024-12-31T20:04:41.181621Z","shell.execute_reply":"2024-12-31T20:04:41.814268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate missing values\nmissing_values = train_data.isnull().mean() * 100\n\n# Plot\nmissing_values.plot(kind='bar', figsize=(10, 6), color='skyblue')\nplt.title('Percentage of Missing Values by Feature')\nplt.ylabel('Percentage')\nplt.xlabel('Features')\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:41.816992Z","iopub.execute_input":"2024-12-31T20:04:41.817392Z","iopub.status.idle":"2024-12-31T20:04:42.832651Z","shell.execute_reply.started":"2024-12-31T20:04:41.817357Z","shell.execute_reply":"2024-12-31T20:04:42.831426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:42.833801Z","iopub.execute_input":"2024-12-31T20:04:42.834107Z","iopub.status.idle":"2024-12-31T20:04:42.855191Z","shell.execute_reply.started":"2024-12-31T20:04:42.834081Z","shell.execute_reply":"2024-12-31T20:04:42.854094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:42.856981Z","iopub.execute_input":"2024-12-31T20:04:42.857482Z","iopub.status.idle":"2024-12-31T20:04:43.017165Z","shell.execute_reply.started":"2024-12-31T20:04:42.857433Z","shell.execute_reply":"2024-12-31T20:04:43.016047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_data.isnull().sum().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:43.018617Z","iopub.execute_input":"2024-12-31T20:04:43.019035Z","iopub.status.idle":"2024-12-31T20:04:43.17364Z","shell.execute_reply.started":"2024-12-31T20:04:43.018998Z","shell.execute_reply":"2024-12-31T20:04:43.172421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate missing values\nmissing_values = original_data.isnull().mean() * 100\n\n# Plot\nmissing_values.plot(kind='bar', figsize=(10, 6), color='skyblue')\nplt.title('Percentage of Missing Values by Feature')\nplt.ylabel('Percentage')\nplt.xlabel('Features')\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:43.17735Z","iopub.execute_input":"2024-12-31T20:04:43.177663Z","iopub.status.idle":"2024-12-31T20:04:43.61372Z","shell.execute_reply.started":"2024-12-31T20:04:43.177639Z","shell.execute_reply":"2024-12-31T20:04:43.612625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_data.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:43.615397Z","iopub.execute_input":"2024-12-31T20:04:43.615823Z","iopub.status.idle":"2024-12-31T20:04:43.774338Z","shell.execute_reply.started":"2024-12-31T20:04:43.615788Z","shell.execute_reply":"2024-12-31T20:04:43.773336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Droping null values:\noriginal_data.dropna(subset=['Premium Amount'],inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:43.775597Z","iopub.execute_input":"2024-12-31T20:04:43.775887Z","iopub.status.idle":"2024-12-31T20:04:43.811168Z","shell.execute_reply.started":"2024-12-31T20:04:43.775861Z","shell.execute_reply":"2024-12-31T20:04:43.81004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Distribution of the data:\noriginal_data.hist(figsize=(10,5),color = 'skyblue', edgecolor='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-11T22:53:01.329461Z","iopub.execute_input":"2024-12-11T22:53:01.329817Z","iopub.status.idle":"2024-12-11T22:53:02.602793Z","shell.execute_reply.started":"2024-12-11T22:53:01.329786Z","shell.execute_reply":"2024-12-11T22:53:02.601582Z"}}},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:43.812633Z","iopub.execute_input":"2024-12-31T20:04:43.812979Z","iopub.status.idle":"2024-12-31T20:04:44.446369Z","shell.execute_reply.started":"2024-12-31T20:04:43.81293Z","shell.execute_reply":"2024-12-31T20:04:44.44532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of the data:\ntrain_data.drop(['id'],axis=1).hist(figsize=(10,5),color = 'skyblue', edgecolor='black')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:44.447718Z","iopub.execute_input":"2024-12-31T20:04:44.448054Z","iopub.status.idle":"2024-12-31T20:04:46.151842Z","shell.execute_reply.started":"2024-12-31T20:04:44.448028Z","shell.execute_reply":"2024-12-31T20:04:46.150744Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Merging original and train data:","metadata":{}},{"cell_type":"code","source":"train_data = train_data.drop(\"id\", axis=1)\n\n#train_data = pd.concat([train_data, original_data], ignore_index=True)\ntrain_data = train_data.drop_duplicates()\nprint(\"shape of the data :\",train_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:46.1532Z","iopub.execute_input":"2024-12-31T20:04:46.15354Z","iopub.status.idle":"2024-12-31T20:04:48.489352Z","shell.execute_reply.started":"2024-12-31T20:04:46.153512Z","shell.execute_reply":"2024-12-31T20:04:48.488306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Policy Start Date'] = pd.to_datetime(train_data['Policy Start Date'])\ntest_data['Policy Start Date'] = pd.to_datetime(test_data['Policy Start Date'])\n\ntrain_data['Policy Start Year'] = train_data['Policy Start Date'].dt.year\ntrain_data['Policy Start Month'] = train_data['Policy Start Date'].dt.month\ntrain_data['Policy Start Day'] = train_data['Policy Start Date'].dt.day\n\ntest_data['Policy Start Year'] = test_data['Policy Start Date'].dt.year\ntest_data['Policy Start Month'] = test_data['Policy Start Date'].dt.month\ntest_data['Policy Start Day'] = test_data['Policy Start Date'].dt.day\n\ntrain_data.drop('Policy Start Date',axis=1,inplace=True)\ntest_data.drop('Policy Start Date',axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:48.490519Z","iopub.execute_input":"2024-12-31T20:04:48.490797Z","iopub.status.idle":"2024-12-31T20:04:49.852938Z","shell.execute_reply.started":"2024-12-31T20:04:48.490775Z","shell.execute_reply":"2024-12-31T20:04:49.851774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col_name in train_data.columns:\n    if train_data[col_name].dtypes=='object':\n        train_data[col_name] = train_data[col_name].fillna(\"missing\")\n    else:\n        train_data[col_name] = train_data[col_name].fillna(train_data[col_name].median())\n        \nfor col_name in test_data.columns:\n    if test_data[col_name].dtypes=='object':\n        test_data[col_name] = test_data[col_name].fillna(\"missing\")\n    else:\n        test_data[col_name] = test_data[col_name].fillna(test_data[col_name].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:49.854367Z","iopub.execute_input":"2024-12-31T20:04:49.85476Z","iopub.status.idle":"2024-12-31T20:04:51.803415Z","shell.execute_reply.started":"2024-12-31T20:04:49.854726Z","shell.execute_reply":"2024-12-31T20:04:51.802311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of the data:\ntrain_data.hist(figsize=(15,10),color = 'skyblue', edgecolor='black')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:51.804825Z","iopub.execute_input":"2024-12-31T20:04:51.805298Z","iopub.status.idle":"2024-12-31T20:04:54.113033Z","shell.execute_reply.started":"2024-12-31T20:04:51.805261Z","shell.execute_reply":"2024-12-31T20:04:54.111874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Premium Amount'] = np.log1p(train_data['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:54.114311Z","iopub.execute_input":"2024-12-31T20:04:54.114626Z","iopub.status.idle":"2024-12-31T20:04:54.144777Z","shell.execute_reply.started":"2024-12-31T20:04:54.114601Z","shell.execute_reply":"2024-12-31T20:04:54.143821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Premium Amount'].hist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:54.146073Z","iopub.execute_input":"2024-12-31T20:04:54.146387Z","iopub.status.idle":"2024-12-31T20:04:54.361863Z","shell.execute_reply.started":"2024-12-31T20:04:54.146362Z","shell.execute_reply":"2024-12-31T20:04:54.360838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_data = train_data.drop('id', axis = 1)\nnum_cols = list(train_data.select_dtypes(exclude=['object']).columns.difference(['Premium Amount']))\ncat_cols = list(train_data.select_dtypes(include=['object']).columns)\n\nnum_cols_test = list(test_data.select_dtypes(exclude=['object']).columns.difference(['id']))\ncat_cols_test = list(test_data.select_dtypes(include=['object']).columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:54.363164Z","iopub.execute_input":"2024-12-31T20:04:54.363472Z","iopub.status.idle":"2024-12-31T20:04:55.378844Z","shell.execute_reply.started":"2024-12-31T20:04:54.363446Z","shell.execute_reply":"2024-12-31T20:04:55.377786Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encoding data:","metadata":{}},{"cell_type":"code","source":"# Convert all categorical columns to string type\n#train_data[cat_cols] = train_data[cat_cols].astype(str)\n#test_data[cat_cols_test] = test_data[cat_cols_test].astype(str)\n# Initialize LabelEncoder\nlabel_encoders = {col: LabelEncoder() for col in cat_cols}\n\n# Apply LabelEncoder to each categorical column\nfor col in cat_cols:\n    train_data[col] = label_encoders[col].fit_transform(train_data[col])\n    test_data[col] = label_encoders[col].transform(test_data[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:55.38058Z","iopub.execute_input":"2024-12-31T20:04:55.380873Z","iopub.status.idle":"2024-12-31T20:04:58.992155Z","shell.execute_reply.started":"2024-12-31T20:04:55.380849Z","shell.execute_reply":"2024-12-31T20:04:58.990981Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Scaling data:","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\ntrain_data[num_cols] = scaler.fit_transform(train_data[num_cols])\ntest_data[num_cols_test] = scaler.transform(test_data[num_cols_test])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:58.993502Z","iopub.execute_input":"2024-12-31T20:04:58.993824Z","iopub.status.idle":"2024-12-31T20:04:59.491943Z","shell.execute_reply.started":"2024-12-31T20:04:58.993797Z","shell.execute_reply":"2024-12-31T20:04:59.49094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = train_data.corr()\n\n# Plot the heatmap\nplt.figure(figsize=(25, 8))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', linewidths=0.5, vmin=-1, vmax=1)\nplt.title('Feature Correlation Heatmap')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:04:59.493237Z","iopub.execute_input":"2024-12-31T20:04:59.493628Z","iopub.status.idle":"2024-12-31T20:05:02.844135Z","shell.execute_reply.started":"2024-12-31T20:04:59.493593Z","shell.execute_reply":"2024-12-31T20:05:02.843147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_data.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:05:02.845609Z","iopub.execute_input":"2024-12-31T20:05:02.846003Z","iopub.status.idle":"2024-12-31T20:05:02.850974Z","shell.execute_reply.started":"2024-12-31T20:05:02.845969Z","shell.execute_reply":"2024-12-31T20:05:02.849816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Splitting Data:","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(['Premium Amount'], axis=1)\ny = train_data['Premium Amount']\ntest = test_data.drop(['id'],axis=1)\n\n# Split datainto training set and test set\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.1, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:05:02.852281Z","iopub.execute_input":"2024-12-31T20:05:02.852623Z","iopub.status.idle":"2024-12-31T20:05:03.636544Z","shell.execute_reply.started":"2024-12-31T20:05:02.852594Z","shell.execute_reply":"2024-12-31T20:05:03.635361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGBMRegressor:  ","metadata":{}},{"cell_type":"code","source":"lgb_params = {'n_estimators': 446, 'max_depth': 0, 'learning_rate': 0.014746128655889696, 'num_leaves': 161, 'min_child_samples': 91, 'subsample': 0.6929832978400176, 'colsample_bytree': 0.9592219644885441, 'reg_alpha': 0.0033550946736107313, 'reg_lambda': 1.4801512847662351e-07}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:05:03.637879Z","iopub.execute_input":"2024-12-31T20:05:03.638232Z","iopub.status.idle":"2024-12-31T20:05:03.643614Z","shell.execute_reply.started":"2024-12-31T20:05:03.638205Z","shell.execute_reply":"2024-12-31T20:05:03.642519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nimport numpy as np\n\n# Define RMSLE metric\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Perform CV with LGBMRegressor\ndef lgbm_cv(X, y, n_splits=5):\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    rmsle_scores = []\n    fold_predictions = []\n    preds = []\n\n    for fold, (train_index, valid_index) in enumerate(kf.split(X)):\n        # Split the data\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n\n        # Initialize and train the model\n        model = lgb.LGBMRegressor(**lgb_params,verbosity=-1,random_state=42)\n        model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)])\n\n        # Predict on validation set\n        y_pred = model.predict(X_valid)\n        fold_predictions.append(y_pred)\n        pred = model.predict(test)\n        preds.append(pred)\n\n        # Calculate RMSLE for this fold\n        score = rmsle(y_valid, y_pred)\n        rmsle_scores.append(score)\n\n        #print(f\"Fold {fold + 1} RMSLE: {score:.4f}\")\n\n    # Compute overall RMSLE\n    mean_rmsle = np.mean(rmsle_scores)\n    print(f\"\\nMean RMSLE: {mean_rmsle:.4f}\")\n\n    return rmsle_scores, preds\n\nrmsle_scores, preds = lgbm_cv(X, y)\n\nlgb_preds = np.mean(np.expm1(preds), axis=0)\nsubmission = pd.DataFrame({'id': test_data.id, 'Premium Amount': lgb_preds})\nprint(submission.head())\nsubmission.to_csv('submission_lgb.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:05:03.64897Z","iopub.execute_input":"2024-12-31T20:05:03.649282Z","iopub.status.idle":"2024-12-31T20:13:35.576038Z","shell.execute_reply.started":"2024-12-31T20:05:03.649258Z","shell.execute_reply":"2024-12-31T20:13:35.57499Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBRegressor:","metadata":{}},{"cell_type":"code","source":"xgb_params = {'n_estimators': 461, 'max_depth': 10, 'learning_rate': 0.017534850635643556, 'subsample': 0.8509295954175136, 'colsample_bytree': 0.9250403118263628, 'min_child_weight': 8, 'reg_alpha': 9.868541080661064, 'reg_lambda': 0.38839011818248154}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:13:35.577392Z","iopub.execute_input":"2024-12-31T20:13:35.577687Z","iopub.status.idle":"2024-12-31T20:13:35.583391Z","shell.execute_reply.started":"2024-12-31T20:13:35.577664Z","shell.execute_reply":"2024-12-31T20:13:35.582141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\n# Initialize K-Fold cross-validator\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Initialize lists to store results\nrmse_scores = []\nrmsle_scores = []\npreds = []\n# Initialize the model\nmodel = XGBRegressor(**xgb_params,loss_function='RMSE')\n\n# K-Fold Cross-Validation\nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n\n    # Fit the model\n    model.fit(X_train, y_train)\n\n    # Predict on the test set\n    y_pred = model.predict(X_test)\n    pred = model.predict(test)\n    preds.append(pred)\n\n    # Calculate RMSE\n    rmse = mean_squared_error(y_test, y_pred, squared=False)\n    rmse_scores.append(rmse)\n    rmsle = mean_squared_log_error(y_test, y_pred, squared=False)\n    rmsle_scores.append(rmsle)\n\n# Calculate the average RMSE across all folds\navg_rmse = np.mean(rmse_scores)\navg_rmsle = np.mean(rmsle_scores)\n\n# Print RMSE for each fold and the average RMSE\nprint(\"Average RMSLE:\", avg_rmsle)\nprint(\"Average RMSE:\", avg_rmse)\n\nxgb_preds = np.mean(np.expm1(preds), axis=0)\nsubmission = pd.DataFrame({'id': test_data.id, 'Premium Amount': xgb_preds})\nprint(submission.head())\nsubmission.to_csv('submission_xgb.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:13:35.58458Z","iopub.execute_input":"2024-12-31T20:13:35.584857Z","iopub.status.idle":"2024-12-31T20:24:06.125001Z","shell.execute_reply.started":"2024-12-31T20:13:35.584834Z","shell.execute_reply":"2024-12-31T20:24:06.123823Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CatBoostRegressor:","metadata":{}},{"cell_type":"code","source":"cat_params = {'iterations': 953, 'depth': 10, 'learning_rate': 0.022243344117282248, 'l2_leaf_reg': 0.8272393843674042, 'bagging_temperature': 0.8849529627304391, 'random_strength': 0.1541145164725649, 'border_count': 247}\n#value: 0.1582848408989202.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:24:06.126527Z","iopub.execute_input":"2024-12-31T20:24:06.12691Z","iopub.status.idle":"2024-12-31T20:24:06.132857Z","shell.execute_reply.started":"2024-12-31T20:24:06.126875Z","shell.execute_reply":"2024-12-31T20:24:06.131526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define RMSLE metric\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Perform CV with CatBoostRegressor\ndef catboost_cv(X, y, n_splits=5):\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    rmsle_scores = []\n    fold_predictions = []\n    preds = []\n\n    for fold, (train_index, valid_index) in enumerate(kf.split(X)):\n        # Split the data\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n\n        # Initialize and train the CatBoost model\n        model = CatBoostRegressor(**cat_params,loss_function=\"RMSE\",random_seed=42,verbose=0)\n        model.fit(X_train, y_train, eval_set=(X_valid, y_valid), early_stopping_rounds=50)\n\n        # Predict on validation set\n        y_pred = model.predict(X_valid)\n        fold_predictions.append(y_pred)\n        pred = model.predict(test)\n        preds.append(pred)\n        \n        # Calculate RMSLE for this fold\n        score = rmsle(y_valid, y_pred)\n        rmsle_scores.append(score)\n\n        #print(f\"Fold {fold + 1} RMSLE: {score:.4f}\")\n\n    # Compute overall RMSLE\n    mean_rmsle = np.mean(rmsle_scores)\n    print(f\"\\nMean RMSLE: {mean_rmsle:.4f}\")\n\n    return rmsle_scores, preds\n\nrmsle_scores, preds = catboost_cv(X, y)\ncat_preds = np.mean(np.expm1(preds), axis=0)\nsubmission = pd.DataFrame({'id': test_data.id, 'Premium Amount': cat_preds})\nprint(submission.head())\nsubmission.to_csv('submission_cat.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:24:06.134175Z","iopub.execute_input":"2024-12-31T20:24:06.134559Z","iopub.status.idle":"2024-12-31T20:39:35.554055Z","shell.execute_reply.started":"2024-12-31T20:24:06.134526Z","shell.execute_reply":"2024-12-31T20:39:35.552983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# HistGradientBoostingRegressor:","metadata":{}},{"cell_type":"code","source":"hgb_params1 = {'learning_rate': 0.02252330514948744, 'max_iter': 650, 'max_leaf_nodes': 85, 'min_samples_leaf': 50, 'l2_regularization': 0.47890735467940704}\nhgb_params = {'learning_rate': 0.03951733038113808, 'max_iter': 400, 'max_leaf_nodes': 80, 'min_samples_leaf': 40, 'l2_regularization': 6.537554383465754e-05}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:39:35.555724Z","iopub.execute_input":"2024-12-31T20:39:35.556139Z","iopub.status.idle":"2024-12-31T20:39:35.561193Z","shell.execute_reply.started":"2024-12-31T20:39:35.556112Z","shell.execute_reply":"2024-12-31T20:39:35.560214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.experimental import enable_hist_gradient_boosting  # Required to use HGBR\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nimport numpy as np\nimport pandas as pd\n\n# Define RMSLE metric\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Perform CV with HistGradientBoostingRegressor\ndef hgbr_cv(X, y, test_data=None, n_splits=5):\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    rmsle_scores = []\n    preds = []\n\n    for fold, (train_index, valid_index) in enumerate(kf.split(X)):\n        # Split the data\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n\n        # Initialize and train the model\n        model = HistGradientBoostingRegressor(**hgb_params,random_state=42)\n        model.fit(X_train, y_train)\n         # Predict on validation set\n        y_pred = model.predict(X_valid)\n\n        # Predict on test data for this fold\n        if test_data is not None:\n            pred = model.predict(test_data)\n            preds.append(pred)\n\n        # Calculate RMSLE for this fold\n        score = rmsle(y_valid, y_pred)\n        rmsle_scores.append(score)\n        print(f\"Fold {fold + 1} RMSLE: {score:.4f}\")\n\n    # Compute overall RMSLE\n    mean_rmsle = np.mean(rmsle_scores)\n    print(f\"\\nMean RMSLE: {mean_rmsle:.4f}\")\n\n    return rmsle_scores, preds\n\nrmsle_scores, preds = hgbr_cv(X, y, test_data=test)\nhgb_preds = np.mean(np.expm1(preds), axis=0)\n\n# Create submission file\nif test_data is not None:\n    submission = pd.DataFrame({\n        'id': data.id,  # Replace with your test data ID column\n        'Premium Amount': hgb_preds  # Reverse log1p if applied\n        })\n    print(submission.head())\n    submission.to_csv('submission_hgbr.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T20:39:35.562725Z","iopub.execute_input":"2024-12-31T20:39:35.563226Z","iopub.status.idle":"2024-12-31T20:43:32.497461Z","shell.execute_reply.started":"2024-12-31T20:39:35.563191Z","shell.execute_reply":"2024-12-31T20:43:32.496237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds = xgb_preds * 0.8 + lgb_preds * 0.1 + cat_preds * 0.05 + hgb_preds * 0.05\n\ndata['Premium Amount'] = test_preds\n\ndata.to_csv('submission_blend.csv', index=False)\n#data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T21:02:43.074731Z","iopub.execute_input":"2024-12-31T21:02:43.075805Z","iopub.status.idle":"2024-12-31T21:02:44.704129Z","shell.execute_reply.started":"2024-12-31T21:02:43.075766Z","shell.execute_reply":"2024-12-31T21:02:44.703017Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"import numpy as np\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\n\n# Define RMSLE metric\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Cross-validation and predictions for a model\ndef get_model_rmsle_and_predictions(model, X, y, test, n_splits=5):\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    rmsle_scores = []\n    test_predictions = np.zeros(len(test))\n    fold_predictions = np.zeros(len(y))\n\n    for train_index, valid_index in kf.split(X):\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n\n        model.fit(X_train, y_train)\n        y_pred = model.predict(X_valid)\n        fold_predictions[valid_index] = y_pred\n        rmsle_scores.append(rmsle(y_valid, y_pred))\n\n        # Predict on the test set for this fold and average\n        test_predictions += model.predict(test) / n_splits\n\n    return np.mean(rmsle_scores), fold_predictions, test_predictions\n\n# Train models and calculate RMSLE\ndef ensemble_predictions(X, y, test):\n    models = {\n        \"LGBM\": LGBMRegressor(**lgb_params,verbosity=-1,random_state=42),\n        \"XGB\": XGBRegressor(**xgb_params,random_state=42),\n        \"CatBoost\": CatBoostRegressor(**cat_params,verbose=0, random_state=42),\n        \"HGB\": HistGradientBoostingRegressor(**hgb_params,random_state=42)\n    }\n    \n    rmsle_scores = {}\n    model_predictions = {}\n    test_predictions = {}\n\n    for name, model in models.items():\n        print(f\"Evaluating {name}...\")\n        score, fold_pred, test_pred = get_model_rmsle_and_predictions(model, X, y, test)\n        rmsle_scores[name] = score\n        model_predictions[name] = fold_pred\n        test_predictions[name] = test_pred\n        print(f\"{name} RMSLE: {score:.4f}\")\n\n    # Compute weights (inverse of RMSLE, normalized)\n    scores_array = np.array(list(rmsle_scores.values()))\n    weights = 1 / scores_array\n    weights /= weights.sum()\n\n    print(\"\\nModel Weights:\")\n    for name, weight in zip(models.keys(), weights):\n        print(f\"{name}: {weight * 100:.2f}%\")\n\n    # Combine predictions using weights\n    final_test_prediction = sum(\n        test_predictions[name] * weight for name, weight in zip(models.keys(), weights)\n    )\n\n    return final_test_prediction, weights\n\nfinal_test_prediction, weights = ensemble_predictions(X, y, test)\n\n# Save final predictions for submission\nsubmission = pd.DataFrame({\n    \"id\": data.id,  # Replace with your test IDs column\n    \"Premium Amount\": np.mean(np.expm1(final_test_prediction), axis=0)  # Replace 'Target' with the appropriate submission column name\n})\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(submission.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-12-31T21:02:58.437628Z","iopub.execute_input":"2024-12-31T21:02:58.438488Z","iopub.status.idle":"2024-12-31T21:12:41.161841Z","shell.execute_reply.started":"2024-12-31T21:02:58.43845Z","shell.execute_reply":"2024-12-31T21:12:41.160052Z"}}},{"cell_type":"markdown","source":"import lightgbm as lgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nimport numpy as np\nimport optuna\n\n# Define RMSLE metric\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Define the objective function for Optuna\ndef objective(trial):\n    # Hyperparameter space for LGBM\n    params = {\n        \"n_estimators\": trial.suggest_int(\"n_estimators\", 50, 500),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 10),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3, log=True),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 300),\n        \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 100),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.6, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.6, 1.0),\n        \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-8, 10.0, log=True),\n        \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-8, 10.0, log=True),\n    }\n\n    # Initialize KFold\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    rmsle_scores = []\n\n    # Cross-validation\n    for train_index, valid_index in kf.split(X):\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n\n        # Initialize LGBM model\n        model = lgb.LGBMRegressor(**params,verbosity=-1, random_state=42)\n        model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], eval_metric=\"rmse\")\n\n        # Predict and calculate RMSLE\n        y_pred = model.predict(X_valid)\n        rmsle_scores.append(rmsle(y_valid, y_pred))\n\n    # Return mean RMSLE score\n    return np.mean(rmsle_scores)\n\n# Create Optuna study\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50)\n\n# Print the best trial\nprint(\"Best RMSLE:\", study.best_value)\nprint(\"Best parameters:\", study.best_params)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-19T02:34:38.625334Z","iopub.execute_input":"2024-12-19T02:34:38.625722Z","iopub.status.idle":"2024-12-19T05:13:45.596934Z","shell.execute_reply.started":"2024-12-19T02:34:38.625689Z","shell.execute_reply":"2024-12-19T05:13:45.595671Z"}}},{"cell_type":"markdown","source":"import optuna\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import cross_val_score, KFold\nfrom sklearn.metrics import make_scorer, mean_squared_log_error\nimport numpy as np\n\n# Define RMSLE as a scoring function\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n# Define the objective function for Optuna\ndef objective(trial):\n    # Suggest hyperparameters to tune\n    params = {\n        \"n_estimators\": trial.suggest_int(\"n_estimators\", 50, 500),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 10),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3, log=True),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.6, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.6, 1.0),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 10),\n        \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-8, 10.0, log=True),\n        \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-8, 10.0, log=True),\n    }\n\n    # Initialize the model\n    model = XGBRegressor(**params, random_state=42, verbosity=0)\n\n    # Cross-validation\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    scores = cross_val_score(\n        model, X, y, cv=kf, scoring=make_scorer(rmsle, greater_is_better=False)\n    )\n\n    # Return the mean RMSLE score\n    return -np.mean(scores)  # Negate because Optuna minimizes the objective\n\n# Create and run the Optuna study\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50)\n\n# Print the best hyperparameters\nprint(\"Best trial:\")\nprint(study.best_trial.params)","metadata":{"_kg_hide-input":true}}]}