{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:08.677804Z","iopub.execute_input":"2024-12-17T00:39:08.678243Z","iopub.status.idle":"2024-12-17T00:39:09.277329Z","shell.execute_reply.started":"2024-12-17T00:39:08.678206Z","shell.execute_reply":"2024-12-17T00:39:09.27581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv').drop(columns='id')\ntest=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv').drop(columns='id')\ntest_id=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:09.278889Z","iopub.execute_input":"2024-12-17T00:39:09.279596Z","iopub.status.idle":"2024-12-17T00:39:28.537965Z","shell.execute_reply.started":"2024-12-17T00:39:09.279545Z","shell.execute_reply":"2024-12-17T00:39:28.536248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:28.539958Z","iopub.execute_input":"2024-12-17T00:39:28.54044Z","iopub.status.idle":"2024-12-17T00:39:28.572471Z","shell.execute_reply.started":"2024-12-17T00:39:28.540387Z","shell.execute_reply":"2024-12-17T00:39:28.571178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nfor i in train.columns:\n    if train[i].dtypes == 'object':  \n        train[i].fillna(train[i].mode()[0], inplace=True) \n    elif train[i].dtypes=='int64':\n    \ttrain[i].fillna(train[i].mean(),inplace=True)\n    elif  train[i].dtypes=='float64':\n    \ttrain[i].fillna(train[i].mean(),inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:28.574164Z","iopub.execute_input":"2024-12-17T00:39:28.574649Z","iopub.status.idle":"2024-12-17T00:39:31.484666Z","shell.execute_reply.started":"2024-12-17T00:39:28.574596Z","shell.execute_reply":"2024-12-17T00:39:31.483452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nfor i in test.columns:\n    if test[i].dtypes == 'object':  \n        test[i].fillna(test[i].mode()[0], inplace=True) \n    elif test[i].dtypes=='int64':\n    \ttest[i].fillna(test[i].mean(),inplace=True)\n    elif  test[i].dtypes=='float64':\n    \ttest[i].fillna(test[i].mean(),inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:31.486203Z","iopub.execute_input":"2024-12-17T00:39:31.486593Z","iopub.status.idle":"2024-12-17T00:39:32.959202Z","shell.execute_reply.started":"2024-12-17T00:39:31.486552Z","shell.execute_reply":"2024-12-17T00:39:32.957963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:32.96359Z","iopub.execute_input":"2024-12-17T00:39:32.963979Z","iopub.status.idle":"2024-12-17T00:39:33.63264Z","shell.execute_reply.started":"2024-12-17T00:39:32.963945Z","shell.execute_reply":"2024-12-17T00:39:33.631552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfor j in train.columns:\n    if train[j].dtypes=='object':\n        train[j]=LabelEncoder().fit_transform(train[j])\n\nfor j in test.columns:\n    if test[j].dtypes=='object':\n        test[j]=LabelEncoder().fit_transform(test[j])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:33.633858Z","iopub.execute_input":"2024-12-17T00:39:33.634159Z","iopub.status.idle":"2024-12-17T00:39:40.641337Z","shell.execute_reply.started":"2024-12-17T00:39:33.634129Z","shell.execute_reply":"2024-12-17T00:39:40.640071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:40.642876Z","iopub.execute_input":"2024-12-17T00:39:40.643719Z","iopub.status.idle":"2024-12-17T00:39:40.664825Z","shell.execute_reply.started":"2024-12-17T00:39:40.643669Z","shell.execute_reply":"2024-12-17T00:39:40.663588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x=train.iloc[:,:-1]\ny=train['Premium Amount']\nx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:40.666777Z","iopub.execute_input":"2024-12-17T00:39:40.667209Z","iopub.status.idle":"2024-12-17T00:39:40.90188Z","shell.execute_reply.started":"2024-12-17T00:39:40.667171Z","shell.execute_reply":"2024-12-17T00:39:40.900726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nxtrain,xtest,ytrain,ytest=train_test_split(x,y,test_size=0.2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:40.903639Z","iopub.execute_input":"2024-12-17T00:39:40.904073Z","iopub.status.idle":"2024-12-17T00:39:41.524453Z","shell.execute_reply.started":"2024-12-17T00:39:40.904024Z","shell.execute_reply":"2024-12-17T00:39:41.523392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xtrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:41.526129Z","iopub.execute_input":"2024-12-17T00:39:41.526775Z","iopub.status.idle":"2024-12-17T00:39:41.547597Z","shell.execute_reply.started":"2024-12-17T00:39:41.526721Z","shell.execute_reply":"2024-12-17T00:39:41.546334Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# AutoGluon","metadata":{}},{"cell_type":"code","source":"pip install autogluon","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:41.549218Z","iopub.execute_input":"2024-12-17T00:39:41.54969Z","iopub.status.idle":"2024-12-17T00:39:59.11068Z","shell.execute_reply.started":"2024-12-17T00:39:41.549639Z","shell.execute_reply":"2024-12-17T00:39:59.109198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ray","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:39:59.112626Z","iopub.execute_input":"2024-12-17T00:39:59.113014Z","iopub.status.idle":"2024-12-17T00:40:10.496301Z","shell.execute_reply.started":"2024-12-17T00:39:59.112978Z","shell.execute_reply":"2024-12-17T00:40:10.494979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip upgrade scikit-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:40:10.498031Z","iopub.execute_input":"2024-12-17T00:40:10.498393Z","iopub.status.idle":"2024-12-17T00:40:12.144328Z","shell.execute_reply.started":"2024-12-17T00:40:10.498355Z","shell.execute_reply":"2024-12-17T00:40:12.142867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import autogluon\nfrom autogluon.tabular import TabularPredictor\n# from sklearn.model_selection import KFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:40:12.146587Z","iopub.execute_input":"2024-12-17T00:40:12.147127Z","iopub.status.idle":"2024-12-17T00:40:12.719258Z","shell.execute_reply.started":"2024-12-17T00:40:12.147071Z","shell.execute_reply":"2024-12-17T00:40:12.718222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# kf = KFold(n_splits=5, shuffle=True)\n# split = kf.split(train, train['Premium Amount'])\n# for i, (_, val_index) in enumerate(split):\n#     train.loc[val_index, 'fold'] = i","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:40:12.720448Z","iopub.execute_input":"2024-12-17T00:40:12.720987Z","iopub.status.idle":"2024-12-17T00:40:12.726078Z","shell.execute_reply.started":"2024-12-17T00:40:12.72095Z","shell.execute_reply":"2024-12-17T00:40:12.724943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictor = TabularPredictor(\n    problem_type='regression',\n    eval_metric='rmse',\n    label='Premium Amount',\n    #groups='fold',\n    verbosity=2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:40:12.727486Z","iopub.execute_input":"2024-12-17T00:40:12.727919Z","iopub.status.idle":"2024-12-17T00:40:12.74102Z","shell.execute_reply.started":"2024-12-17T00:40:12.727872Z","shell.execute_reply":"2024-12-17T00:40:12.739807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictor.fit(\n    train_data=train,\n    time_limit=3600*11,\n    presets='best_quality',\n    excluded_model_types=['KNN']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T00:40:12.742601Z","iopub.execute_input":"2024-12-17T00:40:12.743039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictor.leaderboard(silent=True).style.background_gradient(subset=['score_val'], cmap='RdYlGn')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predict=predictor.predict(test)\npredict.head(5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submit ","metadata":{}},{"cell_type":"code","source":"predict=predictor.predict(test)\n\ndf=pd.DataFrame()\ndf['id']=test_id.astype(int) \ndf['Premium Amount']=predict.astype(int) \nprint(df.head())\ndf.to_csv('ins_sub.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}