{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10264886,"sourceType":"datasetVersion","datasetId":6350398}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:45:50.026781Z","iopub.execute_input":"2024-12-22T11:45:50.027065Z","iopub.status.idle":"2024-12-22T11:45:50.427745Z","shell.execute_reply.started":"2024-12-22T11:45:50.027038Z","shell.execute_reply":"2024-12-22T11:45:50.42632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport lightgbm as lgb\nfrom lightgbm import early_stopping  \nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nfrom sklearn.preprocessing import StandardScaler\n\ntrain1 = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntrain1 = train1.drop(columns=['id'])\n\noriginal = pd.read_csv('/kaggle/input/new-policy-dataset/original.csv')\n\n\n# Drop rows where 'col1' has NaN values\noriginal = original.dropna(subset=['Premium Amount'])\n\n\n\ntrain  = pd.concat([original,train1])\ntrain = train.drop_duplicates()\n\n\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntest_id = test['id']\ntest = test.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:45:50.429134Z","iopub.execute_input":"2024-12-22T11:45:50.42976Z","iopub.status.idle":"2024-12-22T11:46:11.926285Z","shell.execute_reply.started":"2024-12-22T11:45:50.42972Z","shell.execute_reply":"2024-12-22T11:46:11.925171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features = test.select_dtypes(include=['float64']).columns.tolist()\ncategorical_features = test.select_dtypes(include=['object']).columns.tolist()\nprint(categorical_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:46:11.927374Z","iopub.execute_input":"2024-12-22T11:46:11.927685Z","iopub.status.idle":"2024-12-22T11:46:12.102086Z","shell.execute_reply.started":"2024-12-22T11:46:11.927658Z","shell.execute_reply":"2024-12-22T11:46:12.100861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing function for LightGBM\ndef preprocess_lgbm(train, test):\n    # Fill missing values\n    train['Age'] = train['Age'].fillna(train['Age'].mean())\n    train['Annual Income'] = train['Annual Income'].fillna(train['Annual Income'].median())\n    train['Marital Status'] = train['Marital Status'].fillna('Single')\n    train['Number of Dependents'] = train['Number of Dependents'].fillna(train['Number of Dependents'].mean())\n    train['Occupation'] = train['Occupation'].fillna('Unemployed')\n    train['Health Score'] = train['Health Score'].fillna(train['Health Score'].mean())\n    train['Previous Claims'] = train['Previous Claims'].fillna(0.0)\n    train['Vehicle Age'] = train['Vehicle Age'].fillna(train['Vehicle Age'].mean())\n    train['Credit Score'] = train['Credit Score'].fillna(train['Credit Score'].median())\n    train['Insurance Duration'] = train['Insurance Duration'].fillna(train['Insurance Duration'].mean())\n    train['Customer Feedback'] = train['Customer Feedback'].fillna(train['Customer Feedback'].mode()[0])\n\n    test['Age'] = test['Age'].fillna(train['Age'].median())\n    test['Annual Income'] = test['Annual Income'].fillna(test['Annual Income'].median())\n    test['Marital Status'] = test['Marital Status'].fillna('Single')\n    test['Number of Dependents'] = test['Number of Dependents'].fillna(test['Number of Dependents'].mean())\n    test['Occupation'] = test['Occupation'].fillna('Unemployed')\n    test['Health Score'] = test['Health Score'].fillna(test['Health Score'].mean())\n    test['Previous Claims'] = test['Previous Claims'].fillna(0.0)\n    test['Vehicle Age'] = test['Vehicle Age'].fillna(test['Vehicle Age'].mean())\n    test['Credit Score'] = test['Credit Score'].fillna(test['Credit Score'].median())\n    test['Insurance Duration'] = test['Insurance Duration'].fillna(test['Insurance Duration'].mean())\n    test['Customer Feedback'] = test['Customer Feedback'].fillna(test['Customer Feedback'].mode()[0])\n\n    # Handle categorical variables for LGBM\n    #train['Gender'] = train['Gender'] == 'Male'\n    train = pd.get_dummies(train, columns=['Gender','Education Level','Occupation','Location'], drop_first=True)\n    train = pd.get_dummies(train, columns=['Marital Status'], drop_first=True)\n\n    education_mapping = {\"High School\": 1, \"Bachelor's\": 2, \"Master's\": 3, \"PhD\": 4}\n    #train['Education Level'] = train['Education Level'].map(education_mapping)\n\n    occupation_mapping = {'Self-Employed': 1, 'Employed': 1, 'Unemployed': 0}\n    #train['Occupation'] = train['Occupation'].map(occupation_mapping)\n\n    location_mapping = {\"Rural\": 1, \"Suburban\": 2, \"Urban\": 4}\n    #train['Location'] = train['Location'].map(location_mapping)\n\n    # Handle 'Policy Type' column\n    train = pd.get_dummies(train, columns=['Policy Type'], drop_first=True)\n\n    # Handle 'Policy Start Date'\n    train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\n    train['Year'] = train['Policy Start Date'].dt.year\n    train['Day'] = train['Policy Start Date'].dt.day\n    train['Month'] = train['Policy Start Date'].dt.month\n    train[\"Weekday\"] = train[\"Policy Start Date\"].dt.weekday\n    train.drop('Policy Start Date', axis=1, inplace=True)\n\n    # Handle other columns\n    #rating_mapping = {\"Poor\": 1, \"Average\": 2, \"Good\": 3}\n    #train['Customer Feedback'] = train['Customer Feedback'].map(rating_mapping)\n    #train['Smoking Status'] = train['Smoking Status'] == 'Yes'\n    train = pd.get_dummies(train, columns=['Customer Feedback','Smoking Status','Property Type','Exercise Frequency'], drop_first=True)\n    \n    exercise_mapping = {\"Rarely\": 2, \"Monthly\": 1, \"Weekly\": 4, \"Daily\": 16}\n    #train['Exercise Frequency'] = train['Exercise Frequency'].map(exercise_mapping)\n\n    housing_mapping = {\"House\": 1, \"Apartment\": 2, \"Condo\": 4}\n    #train['Property Type'] = train['Property Type'].map(housing_mapping)\n\n    # Preprocessing for test data\n    #test['Gender'] = test['Gender'] == 'Male'\n    test = pd.get_dummies(test, columns=['Gender','Education Level','Occupation','Location'], drop_first=True)\n    test = pd.get_dummies(test, columns=['Marital Status'], drop_first=True)\n    #test['Education Level'] = test['Education Level'].map(education_mapping)\n    #test['Occupation'] = test['Occupation'].map(occupation_mapping)\n    #test['Location'] = test['Location'].map(location_mapping)\n    test = pd.get_dummies(test, columns=['Policy Type'], drop_first=True)\n\n    test['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'])\n    test['Year'] = test['Policy Start Date'].dt.year\n    test['Day'] = test['Policy Start Date'].dt.day\n    test['Month'] = test['Policy Start Date'].dt.month\n    test[\"Weekday\"] = test[\"Policy Start Date\"].dt.weekday\n    test.drop('Policy Start Date', axis=1, inplace=True)\n\n    test = pd.get_dummies(test, columns=['Customer Feedback','Smoking Status','Property Type','Exercise Frequency'], drop_first=True)\n    \n    #test['Customer Feedback'] = test['Customer Feedback'].map(rating_mapping)\n    #test['Smoking Status'] = test['Smoking Status'] == 'Yes'\n    #test['Exercise Frequency'] = test['Exercise Frequency'].map(exercise_mapping)\n    #test['Property Type'] = test['Property Type'].map(housing_mapping)\n\n    return train, test\n\n# Preprocessing function for CatBoost\ndef preprocess_cat(train, test, categorical_features):\n    # Fill missing values for CatBoost (same as above for simplicity)\n    train['Age'] = train['Age'].fillna(train['Age'].mean())\n    train['Annual Income'] = train['Annual Income'].fillna(train['Annual Income'].median())\n    train['Marital Status'] = train['Marital Status'].fillna('Single')\n    train['Number of Dependents'] = train['Number of Dependents'].fillna(train['Number of Dependents'].mean())\n    train['Occupation'] = train['Occupation'].fillna('Unemployed')\n    train['Health Score'] = train['Health Score'].fillna(train['Health Score'].mean())\n    train['Previous Claims'] = train['Previous Claims'].fillna(0.0)\n    train['Vehicle Age'] = train['Vehicle Age'].fillna(train['Vehicle Age'].mean())\n    train['Credit Score'] = train['Credit Score'].fillna(train['Credit Score'].median())\n    train['Insurance Duration'] = train['Insurance Duration'].fillna(train['Insurance Duration'].mean())\n    train['Customer Feedback'] = train['Customer Feedback'].fillna(train['Customer Feedback'].mode()[0])\n\n    test['Age'] = test['Age'].fillna(train['Age'].median())\n    test['Annual Income'] = test['Annual Income'].fillna(test['Annual Income'].median())\n    test['Marital Status'] = test['Marital Status'].fillna('Single')\n    test['Number of Dependents'] = test['Number of Dependents'].fillna(test['Number of Dependents'].mean())\n    test['Occupation'] = test['Occupation'].fillna('Unemployed')\n    test['Health Score'] = test['Health Score'].fillna(test['Health Score'].mean())\n    test['Previous Claims'] = test['Previous Claims'].fillna(0.0)\n    test['Vehicle Age'] = test['Vehicle Age'].fillna(test['Vehicle Age'].mean())\n    test['Credit Score'] = test['Credit Score'].fillna(test['Credit Score'].median())\n    test['Insurance Duration'] = test['Insurance Duration'].fillna(test['Insurance Duration'].mean())\n    test['Customer Feedback'] = test['Customer Feedback'].fillna(test['Customer Feedback'].mode()[0])\n\n    # Convert categorical columns for CatBoost\n    #train['Gender'] = train['Gender'] == 'Male'\n    train['Marital Status'] = train['Marital Status'].fillna('Single')\n    train['Occupation'] = train['Occupation'].fillna('Unemployed')\n\n    #test['Gender'] = test['Gender'] == 'Male'\n    test['Marital Status'] = test['Marital Status'].fillna('Single')\n    test['Occupation'] = test['Occupation'].fillna('Unemployed')\n\n    # Handle categorical features for CatBoost\n    train[categorical_features] = train[categorical_features].astype('category')\n    test[categorical_features] = test[categorical_features].astype('category')\n\n    return train, test\n\n\ntrain_lgbm, test_lgbm = preprocess_lgbm(train.copy(), test.copy())\ntrain_cat, test_cat = preprocess_cat(train.copy(), test.copy(), categorical_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:46:12.103349Z","iopub.execute_input":"2024-12-22T11:46:12.103819Z","iopub.status.idle":"2024-12-22T11:46:23.861796Z","shell.execute_reply.started":"2024-12-22T11:46:12.103769Z","shell.execute_reply":"2024-12-22T11:46:23.860562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X1 = train_lgbm.drop(['Premium Amount'], axis=1)\nX2 = train_cat.drop(['Premium Amount'], axis=1)\n\ny1 = train_lgbm['Premium Amount']\ny2 = train_cat['Premium Amount']\n\nX_test1 = test_lgbm\nX_test2 = test_cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:46:23.864299Z","iopub.execute_input":"2024-12-22T11:46:23.864721Z","iopub.status.idle":"2024-12-22T11:46:24.020012Z","shell.execute_reply.started":"2024-12-22T11:46:23.864692Z","shell.execute_reply":"2024-12-22T11:46:24.018781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nimport numpy as np\nfrom sklearn.model_selection import RepeatedKFold\nfrom catboost import CatBoostRegressor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\n\n# Define RMSLE function\ndef rmsle(y_true, y_pred):\n    y_true = np.array(y_true)\n    y_pred = np.maximum(np.array(y_pred), 0)  # Ensure non-negative predictions\n    return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))\n\n# K-fold cross-validation setup\n\n\n# Model parameters\nparams1 = {'learning_rate': 0.09, 'num_leaves': 100, 'max_depth': 25, 'min_data_in_leaf': 95,\n          'feature_fraction': 0.8, 'bagging_fraction': 0.95, 'bagging_freq': 1,\n          'max_bin': 330, 'min_child_weight': 1, 'scale_pos_weight': 4, 'n_estimators': 250, \"n_jobs\":-1}\n\nparams2 = {\n    'iterations': 500,\n    'learning_rate': 0.05,\n    'depth': 6,\n    'l2_leaf_reg': 3.0,\n    'loss_function': 'RMSE',\n    'verbose': 100  # Output progress every 100 iterations\n}\n\ndef training(params, modelName, X, y, X_test):\n    kfold = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\n    fold_test_preds = []  # List to store test predictions from each fold\n    rmsle_values = []  # List to store RMSLE for each fold\n    \n    for fold, (train_idx, val_idx) in enumerate(tqdm(kfold.split(X, y), desc=\"Training Folds\", total=10)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Apply log transformation to target variables\n        y_train_log = np.log1p(y_train)\n        y_val_log = np.log1p(y_val)\n        \n        # Initialize model based on modelName\n        if modelName == \"lgbm\":\n            model = lgb.LGBMRegressor(**params, random_state=42, verbose=-1, device='cpu')\n        elif modelName == \"catboost\":\n            model = CatBoostRegressor(**params, random_state=42)  # Silent=True for less verbose\n        \n        # Train the model\n        if modelName == \"lgbm\":\n            model.fit(X_train, y_train_log,\n                      eval_set=[(X_val, y_val_log)],\n                      callbacks=[lgb.early_stopping(stopping_rounds=50, verbose=False)])\n        elif modelName == \"catboost\":\n            model.fit(X_train, y_train_log, cat_features=categorical_features, eval_set=(X_val, y_val_log), early_stopping_rounds=50)\n        \n        # Predict on the validation set (log-transformed)\n        y_val_pred_log = model.predict(X_val)\n        y_val_pred = np.expm1(y_val_pred_log)  # Convert predictions back from log scale\n        \n        # Calculate RMSLE for the validation set\n        rmsle_valid = rmsle(y_val, y_val_pred)\n        rmsle_values.append(rmsle_valid)\n        \n        # Predict on the test set\n        test_log_pred = model.predict(X_test)\n        test_pred = np.expm1(test_log_pred)\n        \n        # Append the test predictions for this fold\n        fold_test_preds.append(test_pred)\n        clear_output(wait=True)\n    \n    #print(f\"Average RMSLE across folds: {np.mean(rmsle_values):.5f}\")\n    return fold_test_preds, np.mean(rmsle_values)\n  \n\n#Final test predictions (average over all folds)\nans1 = training(params1,'lgbm', X1, y1, X_test1)\n#ans2  = training(params2, 'catboost', X2, y2, X_test2)\n\nfinal_test_pred1 = np.mean(ans1[0], axis = 0)\nrmsle_values1 = ans1[1]\n#final_test_pred2 = np.mean(ans2[0], axis = 0)\n#rmsle_values2 = ans2[1]\n\nprint(f'LGBM {rmsle_values1}')\n#print(f'Cat {rmsle_values2}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:53:15.130394Z","iopub.execute_input":"2024-12-22T11:53:15.130885Z","iopub.status.idle":"2024-12-22T11:56:04.742183Z","shell.execute_reply.started":"2024-12-22T11:53:15.130851Z","shell.execute_reply":"2024-12-22T11:56:04.740719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = final_test_pred1 #* 0.5 #+ final_test_pred2 * 0.5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:48:50.681057Z","iopub.status.idle":"2024-12-22T11:48:50.681421Z","shell.execute_reply":"2024-12-22T11:48:50.681274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = pd.DataFrame({\n    'id': test_id,\n    'Premium Amount': preds\n})\npredictions.to_csv('predictions5.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T11:48:50.682874Z","iopub.status.idle":"2024-12-22T11:48:50.683389Z","shell.execute_reply":"2024-12-22T11:48:50.683172Z"}},"outputs":[],"execution_count":null}]}