{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:44.173371Z","iopub.execute_input":"2024-12-13T18:13:44.173788Z","iopub.status.idle":"2024-12-13T18:13:44.181991Z","shell.execute_reply.started":"2024-12-13T18:13:44.173751Z","shell.execute_reply":"2024-12-13T18:13:44.180815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport warnings\nwarnings.filterwarnings('ignore')\nimport lightgbm as lgb\nfrom lightgbm import early_stopping  \nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\ntrain = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntrain = train.drop(columns=['id'])\n\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntest_id = test['id']\ntest = test.drop(columns=['id'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:44.184205Z","iopub.execute_input":"2024-12-13T18:13:44.184631Z","iopub.status.idle":"2024-12-13T18:13:51.986447Z","shell.execute_reply.started":"2024-12-13T18:13:44.184579Z","shell.execute_reply":"2024-12-13T18:13:51.98546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:51.987634Z","iopub.execute_input":"2024-12-13T18:13:51.987913Z","iopub.status.idle":"2024-12-13T18:13:52.63627Z","shell.execute_reply.started":"2024-12-13T18:13:51.987885Z","shell.execute_reply":"2024-12-13T18:13:52.635121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Age'] = train['Age'].fillna(train['Age'].mean())\ntrain['Annual Income'] = train['Annual Income'].fillna(train['Annual Income'].mean())\ntrain['Marital Status'] = train['Marital Status'].fillna('Single')\ntrain['Number of Dependents'] = train['Number of Dependents'].fillna(0.0)\ntrain['Occupation'] = train['Occupation'].fillna('Unemployed')\ntrain['Health Score'] = train['Health Score'].fillna(train['Health Score'].mean())\ntrain['Previous Claims'] = train['Previous Claims'].fillna(0.0)\ntrain['Vehicle Age'] = train['Vehicle Age'].fillna(train['Vehicle Age'].mean())\ntrain['Credit Score'] = train['Credit Score'].fillna(train['Credit Score'].mean())\ntrain['Insurance Duration'] = train['Insurance Duration'].fillna(train['Insurance Duration'].mean())\ntrain['Customer Feedback'] = train['Customer Feedback'].fillna(train['Customer Feedback'].mode()[0])\n\ntest['Age'] = test['Age'].fillna(train['Age'].mean())\ntest['Annual Income'] = test['Annual Income'].fillna(test['Annual Income'].mean())\ntest['Marital Status'] = test['Marital Status'].fillna('Single')\ntest['Number of Dependents'] = test['Number of Dependents'].fillna(0.0)\ntest['Occupation'] = test['Occupation'].fillna('Unemployed')\ntest['Health Score'] = test['Health Score'].fillna(test['Health Score'].mean())\ntest['Previous Claims'] = test['Previous Claims'].fillna(0.0)\ntest['Vehicle Age'] = test['Vehicle Age'].fillna(test['Vehicle Age'].mean())\ntest['Credit Score'] = test['Credit Score'].fillna(test['Credit Score'].mean())\ntest['Insurance Duration'] = test['Insurance Duration'].fillna(test['Insurance Duration'].mean())\ntest['Customer Feedback'] = test['Customer Feedback'].fillna(test['Customer Feedback'].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:52.637565Z","iopub.execute_input":"2024-12-13T18:13:52.637894Z","iopub.status.idle":"2024-12-13T18:13:53.369967Z","shell.execute_reply.started":"2024-12-13T18:13:52.637862Z","shell.execute_reply":"2024-12-13T18:13:53.368543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Gender'] = train['Gender'] == 'Male'\ntrain = pd.get_dummies(train, columns=['Marital Status'], drop_first = True)\neducation_mapping = {\n    \"High School\": 1,\n    \"Bachelor's\": 2,\n    \"Master's\": 3,\n    \"PhD\": 4\n}\ntrain['Education Level'] = train['Education Level'].map(education_mapping)\ntrain = pd.get_dummies(train, columns=['Occupation'], drop_first = True)\n\nlocation_mapping = {\n    \"Rural\": 1,\n    \"Suburban\": 2,\n    \"Urabn\": 3\n}\n\ntrain = pd.get_dummies(train, columns=['Location'], drop_first = True)\ntrain = pd.get_dummies(train, columns=['Policy Type'], drop_first = True)\n# train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\n# train['Policy Start Date'] = train['Policy Start Date'].view('int64')\n\ntrain['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\ntrain['Year'] = train['Policy Start Date'].dt.year\ntrain['Day'] = train['Policy Start Date'].dt.day\ntrain['Month'] = train['Policy Start Date'].dt.month\ntrain[\"Weekday\"] = train[\"Policy Start Date\"].dt.weekday\ntrain.drop('Policy Start Date', axis=1, inplace=True)\n\nrating_mapping = {\n    \"Poor\": 1,\n    \"Average\": 2,\n    \"Good\": 3\n}\ntrain['Customer Feedback'] = train['Customer Feedback'].map(rating_mapping)\ntrain['Smoking Status'] = train['Smoking Status'] == 'Yes'\n\nexercise_mapping = {\n    \"Rarely\": 1,\n    \"Monthly\": 2,\n    \"Weekly\": 4,\n    \"Daily\": 8\n}\ntrain['Exercise Frequency'] = train['Exercise Frequency'].map(exercise_mapping)\n\nhousing_mapping = {\n    \"House\": 1,\n    \"Apartment\": 2,\n    \"Condo\": 4,\n}\ntrain['Property Type'] = train['Property Type'].map(housing_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:53.372616Z","iopub.execute_input":"2024-12-13T18:13:53.372971Z","iopub.status.idle":"2024-12-13T18:13:56.915375Z","shell.execute_reply.started":"2024-12-13T18:13:53.372936Z","shell.execute_reply":"2024-12-13T18:13:56.914137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['Gender'] = test['Gender'] == 'Male'\ntest = pd.get_dummies(test, columns=['Marital Status'], drop_first = True)\ntest['Education Level'] = test['Education Level'].map(education_mapping)\ntest = pd.get_dummies(test, columns=['Occupation'], drop_first = True)\ntest = pd.get_dummies(test, columns=['Location'], drop_first = True)\ntest = pd.get_dummies(test, columns=['Policy Type'], drop_first = True)\n# test['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\n# test['Policy Start Date'] = test['Policy Start Date'].view('int64')\n\ntest['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'])\ntest['Year'] = test['Policy Start Date'].dt.year\ntest['Day'] = test['Policy Start Date'].dt.day\ntest['Month'] = test['Policy Start Date'].dt.month\ntest[\"Weekday\"] = test[\"Policy Start Date\"].dt.weekday\ntest.drop('Policy Start Date', axis=1, inplace=True)\n\ntest['Customer Feedback'] = test['Customer Feedback'].map(rating_mapping)\ntest['Smoking Status'] = test['Smoking Status'] == 'Yes'\ntest['Exercise Frequency'] = test['Exercise Frequency'].map(exercise_mapping)\ntest['Property Type'] = test['Property Type'].map(housing_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:56.916838Z","iopub.execute_input":"2024-12-13T18:13:56.917194Z","iopub.status.idle":"2024-12-13T18:13:59.155242Z","shell.execute_reply.started":"2024-12-13T18:13:56.917161Z","shell.execute_reply":"2024-12-13T18:13:59.154074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop(['Premium Amount'], axis=1)\ny = train['Premium Amount']\nX_test = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:59.156745Z","iopub.execute_input":"2024-12-13T18:13:59.157213Z","iopub.status.idle":"2024-12-13T18:13:59.231352Z","shell.execute_reply.started":"2024-12-13T18:13:59.157161Z","shell.execute_reply":"2024-12-13T18:13:59.23016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nkfold = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\nfold_test_preds = []\nparams = {'learning_rate': 0.09, 'num_leaves': 100, 'max_depth': 25, 'min_data_in_leaf': 95,\n          'feature_fraction': 0.8, 'bagging_fraction': 0.95, 'bagging_freq': 1,\n          'max_bin': 330, 'min_child_weight': 1, 'scale_pos_weight': 4,'n_estimators':250}\nfor fold, (train_idx, val_idx) in enumerate(tqdm(kfold.split(X, y), desc=\"Training Folds\", total=10)):\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n    y_train_log = np.log1p(y_train)\n    y_val_log = np.log1p(y_val)\n        \n    model = lgb.LGBMRegressor(**params, random_state=42, verbose=-1, n_jobs=-1, device='cpu')\n    model.fit(X_train, y_train_log,\n              eval_set=[(X_val, y_val_log)],\n              callbacks=[lgb.early_stopping(stopping_rounds=50, verbose=False)])\n\n    test_log_pred = model.predict(X_test)\n    test_pred = np.expm1(test_log_pred)\n\n    fold_test_preds.append(test_pred)\n    clear_output(wait=True)\n    \nmean_test_preds = np.mean(fold_test_preds, axis=0)\npreds = mean_test_preds[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:13:59.232682Z","iopub.execute_input":"2024-12-13T18:13:59.233048Z","iopub.status.idle":"2024-12-13T18:18:08.012749Z","shell.execute_reply.started":"2024-12-13T18:13:59.232996Z","shell.execute_reply":"2024-12-13T18:18:08.011538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = pd.DataFrame({\n    'id': test_id,\n    'Premium Amount': preds\n})\npredictions.to_csv('ans.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T18:18:08.014305Z","iopub.execute_input":"2024-12-13T18:18:08.014622Z","iopub.status.idle":"2024-12-13T18:18:09.719594Z","shell.execute_reply.started":"2024-12-13T18:18:08.014592Z","shell.execute_reply":"2024-12-13T18:18:09.718344Z"}},"outputs":[],"execution_count":null}]}