{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:13.420549Z","iopub.execute_input":"2024-12-24T11:50:13.420894Z","iopub.status.idle":"2024-12-24T11:50:13.791773Z","shell.execute_reply.started":"2024-12-24T11:50:13.420867Z","shell.execute_reply":"2024-12-24T11:50:13.790304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nid_test = test['id']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:13.792988Z","iopub.execute_input":"2024-12-24T11:50:13.793362Z","iopub.status.idle":"2024-12-24T11:50:24.014482Z","shell.execute_reply.started":"2024-12-24T11:50:13.793336Z","shell.execute_reply":"2024-12-24T11:50:24.013442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:24.016137Z","iopub.execute_input":"2024-12-24T11:50:24.016506Z","iopub.status.idle":"2024-12-24T11:50:24.054106Z","shell.execute_reply.started":"2024-12-24T11:50:24.016466Z","shell.execute_reply":"2024-12-24T11:50:24.053053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:24.055682Z","iopub.execute_input":"2024-12-24T11:50:24.056051Z","iopub.status.idle":"2024-12-24T11:50:24.076263Z","shell.execute_reply.started":"2024-12-24T11:50:24.056016Z","shell.execute_reply":"2024-12-24T11:50:24.07534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:24.077168Z","iopub.execute_input":"2024-12-24T11:50:24.077499Z","iopub.status.idle":"2024-12-24T11:50:24.095801Z","shell.execute_reply.started":"2024-12-24T11:50:24.077437Z","shell.execute_reply":"2024-12-24T11:50:24.09475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##날짜 DateTime 만들어주기 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:24.096863Z","iopub.execute_input":"2024-12-24T11:50:24.097304Z","iopub.status.idle":"2024-12-24T11:50:24.115165Z","shell.execute_reply.started":"2024-12-24T11:50:24.097266Z","shell.execute_reply":"2024-12-24T11:50:24.113954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_datetime(df):\n    df = df.drop('id',axis = 1)\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Policy_Start_Year'] = df['Policy Start Date'].dt.year\n    df['Policy_Start_Month'] = df['Policy Start Date'].dt.month\n    df['Policy_Start_Day'] = df['Policy Start Date'].dt.day\n    return df\ntrain = to_datetime(train)\ntest = to_datetime(test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:24.116124Z","iopub.execute_input":"2024-12-24T11:50:24.11661Z","iopub.status.idle":"2024-12-24T11:50:25.433012Z","shell.execute_reply.started":"2024-12-24T11:50:24.116562Z","shell.execute_reply":"2024-12-24T11:50:25.432054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:25.436864Z","iopub.execute_input":"2024-12-24T11:50:25.437139Z","iopub.status.idle":"2024-12-24T11:50:26.03861Z","shell.execute_reply.started":"2024-12-24T11:50:25.437115Z","shell.execute_reply":"2024-12-24T11:50:26.037362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##결측값 처리 처음엔 평균으로 했다가 자꾸 점수 안나와서 중간값으로 바꿈 ㅠㅠ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:26.040101Z","iopub.execute_input":"2024-12-24T11:50:26.040349Z","iopub.status.idle":"2024-12-24T11:50:26.04442Z","shell.execute_reply.started":"2024-12-24T11:50:26.040327Z","shell.execute_reply":"2024-12-24T11:50:26.043436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## 결측값 처리 ()\n\ndef float_missing_data(df):\n    float = df.select_dtypes(exclude = 'object').columns\n    for data in float:\n        df[data] = df[data].fillna(-1)\n    return df\n\ndef object_missing_data(df):\n    objec = df.select_dtypes(include = 'object').columns\n    for data in objec:\n        df[data] = df[data].fillna(\"Unknown\")\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:26.045752Z","iopub.execute_input":"2024-12-24T11:50:26.046096Z","iopub.status.idle":"2024-12-24T11:50:26.062342Z","shell.execute_reply.started":"2024-12-24T11:50:26.046062Z","shell.execute_reply":"2024-12-24T11:50:26.061426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = float_missing_data(train)\ntrain = object_missing_data(train)\n\ntest = float_missing_data(test)\ntest = object_missing_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:26.063708Z","iopub.execute_input":"2024-12-24T11:50:26.064113Z","iopub.status.idle":"2024-12-24T11:50:28.429786Z","shell.execute_reply.started":"2024-12-24T11:50:26.064053Z","shell.execute_reply":"2024-12-24T11:50:28.428688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## 결측값 처리 잘됐다\n\ntrain.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:28.430764Z","iopub.execute_input":"2024-12-24T11:50:28.431036Z","iopub.status.idle":"2024-12-24T11:50:29.023314Z","shell.execute_reply.started":"2024-12-24T11:50:28.431012Z","shell.execute_reply":"2024-12-24T11:50:29.022397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##인코딩~\n\ngender = {\"Female\": 1 , \"Male\" : 2, \"Unknown\" : 0}\nmarital_status = {\"Single\" :1 , \"Divorced\":2,\"Married\":3, \"Unknown\" : 0}\neducation_level = {\"High School\":1 , \"Bachelor's\":2 , \"Master\" : 3 , \"PhD\":4  , \"Unknown\" : 0}\noccupation = {\"Unemployed\" : 1 , \"Employed\":2, \"Self-Employed\" : 3 , \"Unknown\" : 0}\nlocation = {\"Urban\" : 1 , \"Rural\" : 2 , \"Suburban\" : 3, \"Unknown\" : 0}\npolicy = {\"Basic\" : 1 , \"Comprehensive\" : 2 , \"Premium\" : 3, \"Unknown\" : 0}\nfeedback = {\"Pool\" : 1 , \"Average\" : 2 , \"Good\" : 3, \"Unknown\" : 0}\nsmoke = {\"No\" : 1, \"Yes\":2,\"Unknown\" : 0}\nexercise = {\"Rarely\" : 1 , \"Daily\" : 2, \"Weekly\" : 3 , \"Monthly\" : 4, \"Unknown\" : 0}\nproperty = {\"House\" : 1 , \"Apartment\" : 2 , \"Condo\" : 3, \"Unknown\" : 0}\n\ndef mapping(df):\n    df['Gender'] = df['Gender'].map(gender)\n    df['Marital Status'] = df['Marital Status'].map(marital_status)\n    df['Education Level'] = df['Education Level'].map(education_level)\n    df['Occupation'] = df['Occupation'].map(occupation)\n    df['Location'] = df['Location'].map(location)\n    df['Policy Type'] = df['Policy Type'].map(policy)\n    df['Customer Feedback'] = df['Customer Feedback'].map(feedback)\n    df['Smoking Status'] = df['Smoking Status'].map(smoke)\n    df['Exercise Frequency'] = df['Exercise Frequency'].map(exercise)\n    df['Property Type'] = df['Property Type'].map(property)\n    return df\n\ntrain = mapping(train)\ntest = mapping(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:29.024228Z","iopub.execute_input":"2024-12-24T11:50:29.024515Z","iopub.status.idle":"2024-12-24T11:50:30.275448Z","shell.execute_reply.started":"2024-12-24T11:50:29.024492Z","shell.execute_reply":"2024-12-24T11:50:30.274291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## 괜히 한번 더 돌려줘야됨 ㅡㅡ\ntrain = float_missing_data(train)\ntrain = object_missing_data(train)\n\ntest = float_missing_data(test)\ntest = object_missing_data(test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:30.276564Z","iopub.execute_input":"2024-12-24T11:50:30.276916Z","iopub.status.idle":"2024-12-24T11:50:30.811354Z","shell.execute_reply.started":"2024-12-24T11:50:30.27688Z","shell.execute_reply":"2024-12-24T11:50:30.810546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#쓰는 것만 쓰자.. 시간만 오래 걸리게\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:30.812322Z","iopub.execute_input":"2024-12-24T11:50:30.812697Z","iopub.status.idle":"2024-12-24T11:50:31.408913Z","shell.execute_reply.started":"2024-12-24T11:50:30.81266Z","shell.execute_reply":"2024-12-24T11:50:31.407604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 이거 드랍하는거 까먹음\ntrain = train.drop('Policy Start Date',axis = 1)\ntest = test.drop('Policy Start Date',axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:31.410022Z","iopub.execute_input":"2024-12-24T11:50:31.410537Z","iopub.status.idle":"2024-12-24T11:50:31.589526Z","shell.execute_reply.started":"2024-12-24T11:50:31.410502Z","shell.execute_reply":"2024-12-24T11:50:31.588364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmsle(y_true,y_pred):\n    log_true = np.log1p(y_true)\n    log_pred = np.log1p(y_pred)\n    return np.sqrt(mean_squared_error(log_true,log_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:31.590531Z","iopub.execute_input":"2024-12-24T11:50:31.590787Z","iopub.status.idle":"2024-12-24T11:50:31.595378Z","shell.execute_reply.started":"2024-12-24T11:50:31.590766Z","shell.execute_reply":"2024-12-24T11:50:31.594274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = train['Premium Amount']\nx = train.drop('Premium Amount',axis = 1)\ny_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:31.596357Z","iopub.execute_input":"2024-12-24T11:50:31.596701Z","iopub.status.idle":"2024-12-24T11:50:31.706873Z","shell.execute_reply.started":"2024-12-24T11:50:31.596677Z","shell.execute_reply":"2024-12-24T11:50:31.705978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train , x_val , y_train , y_val = train_test_split(x,y_log,test_size = 0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:31.707806Z","iopub.execute_input":"2024-12-24T11:50:31.708149Z","iopub.status.idle":"2024-12-24T11:50:32.183045Z","shell.execute_reply.started":"2024-12-24T11:50:31.708116Z","shell.execute_reply":"2024-12-24T11:50:32.182083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\n\nxgb_model = xgb.XGBRegressor(verbosity = 0 , device = 'gpu',n_estimators = 1000,\n                             learning_rate = 0.01,\n                             objective = 'reg:squarederror', gamma = 0.3)\nxgb_model.fit(x_train,y_train,eval_set = [(x_val,y_val)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:50:32.183933Z","iopub.execute_input":"2024-12-24T11:50:32.184186Z","iopub.status.idle":"2024-12-24T11:52:06.703495Z","shell.execute_reply.started":"2024-12-24T11:50:32.184162Z","shell.execute_reply":"2024-12-24T11:52:06.702748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_xgb = np.expm1(xgb_model.predict(x_val))\ny_val_true = np.expm1(y_val)\ntrain_min = y_train.min()\nclipped_predictions = np.maximum(pred_xgb, train_min)\n\nxgb_rmsle = rmsle(y_val_true, pred_xgb)\nxgb_rmsle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:06.704589Z","iopub.execute_input":"2024-12-24T11:52:06.70492Z","iopub.status.idle":"2024-12-24T11:52:09.033486Z","shell.execute_reply.started":"2024-12-24T11:52:06.704895Z","shell.execute_reply":"2024-12-24T11:52:09.032411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as light\n\nlight_model = light.LGBMRegressor(learning_rate = 0.2, n_estimators = 500, verbose=0)\nlight_model.fit(x_train, y_train, eval_set=[(x_val, y_val)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:09.034415Z","iopub.execute_input":"2024-12-24T11:52:09.034689Z","iopub.status.idle":"2024-12-24T11:52:30.271635Z","shell.execute_reply.started":"2024-12-24T11:52:09.034665Z","shell.execute_reply":"2024-12-24T11:52:30.27059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_light = np.expm1(light_model.predict(x_val))\ny_val_true = np.expm1(y_val)\ntrain_min = y_train.min()\nclipped_predictions = np.maximum(pred_light, train_min)\n\nlight_gbm_rmsle = rmsle(y_val_true, pred_light)\nlight_gbm_rmsle\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:30.275485Z","iopub.execute_input":"2024-12-24T11:52:30.276087Z","iopub.status.idle":"2024-12-24T11:52:32.572124Z","shell.execute_reply.started":"2024-12-24T11:52:30.276057Z","shell.execute_reply":"2024-12-24T11:52:32.571021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingRegressor as hgbc\n\nhgbc_model = hgbc(l2_regularization = 2 , learning_rate = 0.01 , max_iter = 300)\nhgbc_model.fit(x_train , y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:32.574064Z","iopub.execute_input":"2024-12-24T11:52:32.574488Z","iopub.status.idle":"2024-12-24T11:52:50.642359Z","shell.execute_reply.started":"2024-12-24T11:52:32.574431Z","shell.execute_reply":"2024-12-24T11:52:50.640787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = np.expm1(hgbc_model.predict(x_val))\ny_val_true = np.expm1(y_val)\ntrain_min = y_train.min()\nclipped_predictions = np.maximum(pred , train_min)\n\nhgbc_rmsle = rmsle(y_val_true,pred)\nhgbc_rmsle\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:50.642955Z","iopub.execute_input":"2024-12-24T11:52:50.643213Z","iopub.status.idle":"2024-12-24T11:52:53.282698Z","shell.execute_reply.started":"2024-12-24T11:52:50.643172Z","shell.execute_reply":"2024-12-24T11:52:53.281567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = {\"XGBRegressor\" : xgb_rmsle,\n        \"LightGBMRegressor\" : light_gbm_rmsle,\n        \"HyperGBCRegressor\" : hgbc_rmsle}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:53.283758Z","iopub.execute_input":"2024-12-24T11:52:53.284422Z","iopub.status.idle":"2024-12-24T11:52:53.288818Z","shell.execute_reply.started":"2024-12-24T11:52:53.284388Z","shell.execute_reply":"2024-12-24T11:52:53.287761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:53.289807Z","iopub.execute_input":"2024-12-24T11:52:53.29018Z","iopub.status.idle":"2024-12-24T11:52:53.30942Z","shell.execute_reply.started":"2024-12-24T11:52:53.290148Z","shell.execute_reply":"2024-12-24T11:52:53.308337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted = hgbc_model.predict(test)\nsubmission = pd.DataFrame({'id': id_test , 'Premium Amount' : np.expm1(predicted)})\nsubmission.to_csv(\"submission.csv\",index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T11:52:53.310354Z","iopub.execute_input":"2024-12-24T11:52:53.310688Z","iopub.status.idle":"2024-12-24T11:53:03.245549Z","shell.execute_reply.started":"2024-12-24T11:52:53.310663Z","shell.execute_reply":"2024-12-24T11:53:03.244574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}