{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport os\nimport joblib\nimport numpy as np\nimport pandas as pd\n\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import KFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-13T06:29:10.480192Z","iopub.execute_input":"2022-12-13T06:29:10.480726Z","iopub.status.idle":"2022-12-13T06:29:12.644376Z","shell.execute_reply.started":"2022-12-13T06:29:10.480607Z","shell.execute_reply":"2022-12-13T06:29:12.64292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(f'/kaggle/input/g2net-exploring-the-mysterious-eda/generated_data.csv')\n#train = train[train['label']!=-1]\ntrain.reset_index(inplace = True, drop = True)\ntrain.drop(\"Unnamed: 0\",inplace=True,axis=1)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:08.954856Z","iopub.execute_input":"2022-12-13T06:32:08.95604Z","iopub.status.idle":"2022-12-13T06:32:08.998324Z","shell.execute_reply.started":"2022-12-13T06:32:08.955988Z","shell.execute_reply":"2022-12-13T06:32:08.997328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:12.710906Z","iopub.execute_input":"2022-12-13T06:32:12.711286Z","iopub.status.idle":"2022-12-13T06:32:12.720099Z","shell.execute_reply.started":"2022-12-13T06:32:12.711253Z","shell.execute_reply":"2022-12-13T06:32:12.719151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lgbm = LGBMClassifier(n_estimators = 40000)\nmodel_cat = CatBoostClassifier(iterations=25, learning_rate=0.5)\nmodel_xgb = XGBClassifier()\n# model_svm = SVC()\nmodel_rf = RandomForestClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:13.559236Z","iopub.execute_input":"2022-12-13T06:32:13.560245Z","iopub.status.idle":"2022-12-13T06:32:13.565635Z","shell.execute_reply.started":"2022-12-13T06:32:13.560201Z","shell.execute_reply":"2022-12-13T06:32:13.56449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [i for i in train.columns if i not in ['label']]\nX_train, X_test, y_train, y_test = train_test_split(train[FEATURES], train['label'], test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:21.436817Z","iopub.execute_input":"2022-12-13T06:32:21.437235Z","iopub.status.idle":"2022-12-13T06:32:21.44596Z","shell.execute_reply.started":"2022-12-13T06:32:21.437203Z","shell.execute_reply":"2022-12-13T06:32:21.444736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lgbm.fit(X_train,y_train,eval_set=[(X_test,y_test)],early_stopping_rounds=100, verbose=200)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:24.693859Z","iopub.execute_input":"2022-12-13T06:32:24.694252Z","iopub.status.idle":"2022-12-13T06:32:24.894655Z","shell.execute_reply.started":"2022-12-13T06:32:24.69422Z","shell.execute_reply":"2022-12-13T06:32:24.893655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel_cat.fit(X_train,y_train,eval_set=[(X_test,y_test)])","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:25.125059Z","iopub.execute_input":"2022-12-13T06:32:25.126352Z","iopub.status.idle":"2022-12-13T06:32:25.335305Z","shell.execute_reply.started":"2022-12-13T06:32:25.126304Z","shell.execute_reply":"2022-12-13T06:32:25.333291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_xgb.fit(X_train,y_train,eval_set=[(X_test,y_test)])","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:27.296473Z","iopub.execute_input":"2022-12-13T06:32:27.297334Z","iopub.status.idle":"2022-12-13T06:32:27.991446Z","shell.execute_reply.started":"2022-12-13T06:32:27.297292Z","shell.execute_reply":"2022-12-13T06:32:27.990451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_rf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:32:30.521753Z","iopub.execute_input":"2022-12-13T06:32:30.523086Z","iopub.status.idle":"2022-12-13T06:32:30.762512Z","shell.execute_reply.started":"2022-12-13T06:32:30.523034Z","shell.execute_reply":"2022-12-13T06:32:30.761288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(f'/kaggle/input/g2net-exploring-the-mysterious-eda/test_submission.csv')\nfor column in test.columns:\n    if column not in [\"id\",'label']:\n        test[column] = test[column].apply(lambda x:np.absolute(complex(x)))","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:49:06.581482Z","iopub.execute_input":"2022-12-13T06:49:06.581916Z","iopub.status.idle":"2022-12-13T06:49:07.098204Z","shell.execute_reply.started":"2022-12-13T06:49:06.581877Z","shell.execute_reply":"2022-12-13T06:49:07.096985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:49:31.704945Z","iopub.execute_input":"2022-12-13T06:49:31.70534Z","iopub.status.idle":"2022-12-13T06:49:31.713187Z","shell.execute_reply.started":"2022-12-13T06:49:31.705308Z","shell.execute_reply":"2022-12-13T06:49:31.711836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\ntest_indexs = []\n\nfor j, model in enumerate([model_rf, model_lgbm, model_cat, model_xgb]):\n    preds = []\n#     for i in range(1,81):\n    #test = pd.read_csv(f'/kaggle/input/g2net-exploring-the-mysterious-eda/test_submission.csv')\n    preds += [i[1] for i in model.predict_proba(test[FEATURES])]\n    if(j == 0):\n        test_indexs += test['id'].tolist()\n    #del test\n    gc.collect()\n    predictions.append(np.array(preds))\n    del preds\n    gc.collect()\n\npredictions = np.average(predictions, axis = 0)\nsub = pd.DataFrame(list(zip(test_indexs, predictions)), columns = ['id','target'])\nsub = sub.groupby('id').mean().reset_index()\nsub.to_csv('submission.csv', index = False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:49:08.083375Z","iopub.execute_input":"2022-12-13T06:49:08.084615Z","iopub.status.idle":"2022-12-13T06:49:11.234677Z","shell.execute_reply.started":"2022-12-13T06:49:08.084569Z","shell.execute_reply":"2022-12-13T06:49:11.233323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}