{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-10T14:15:52.215732Z","iopub.execute_input":"2023-01-10T14:15:52.216174Z","iopub.status.idle":"2023-01-10T14:16:21.371975Z","shell.execute_reply.started":"2023-01-10T14:15:52.216143Z","shell.execute_reply":"2023-01-10T14:16:21.370978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"⭐️⭐️RSNA Screening Mammograpy Breast Cancer Detection⭐️⭐️\nSteps:\n1. Import Necessary Library\n2. Load and analysis the data\n3. Preprocessing\n4. KFOLD\n5. Build the Model\n6. Predict Output\n7. Generate Submission file","metadata":{}},{"cell_type":"markdown","source":"# importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport cv2\nimport missingno as msno\nimport plotly.graph_objs as go\nimport plotly.express as px\n\nfrom wordcloud import WordCloud, STOPWORDS\n#Preprocessing\nfrom sklearn import model_selection\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder,StandardScaler\nfrom sklearn import preprocessing\nfrom sklearn.impute import KNNImputer\nfrom sklearn.impute import SimpleImputer\n#Model\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:21.374089Z","iopub.execute_input":"2023-01-10T14:16:21.374766Z","iopub.status.idle":"2023-01-10T14:16:24.381845Z","shell.execute_reply.started":"2023-01-10T14:16:21.374724Z","shell.execute_reply":"2023-01-10T14:16:24.380875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\nsample = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\ntrain_image_path = '/kaggle/input/rsna-breast-cancer-detection/train_images'\ntest_image_path = '/kaggle/input/rsna-breast-cancer-detection/test_images'","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.383157Z","iopub.execute_input":"2023-01-10T14:16:24.383525Z","iopub.status.idle":"2023-01-10T14:16:24.492786Z","shell.execute_reply.started":"2023-01-10T14:16:24.383491Z","shell.execute_reply":"2023-01-10T14:16:24.491792Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train_Shape: {train.shape},Test_Shape: {test.shape},Sample_Shape: {sample.shape}')\ndisplay(train.sample(2))\ndisplay(test.sample(2))\ndisplay(sample.sample(2))","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.495471Z","iopub.execute_input":"2023-01-10T14:16:24.495864Z","iopub.status.idle":"2023-01-10T14:16:24.532352Z","shell.execute_reply.started":"2023-01-10T14:16:24.495812Z","shell.execute_reply":"2023-01-10T14:16:24.531307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.533887Z","iopub.execute_input":"2023-01-10T14:16:24.534313Z","iopub.status.idle":"2023-01-10T14:16:24.563322Z","shell.execute_reply.started":"2023-01-10T14:16:24.534278Z","shell.execute_reply":"2023-01-10T14:16:24.562248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Description\n\n* [train/test]_images/[patient_id]/[image_id].dcm The mammograms, in dicom format. You can expect roughly 8,000 patients in the hidden test set. There are usually but not always 4 images per patient. Note that many of the images use the jpeg 2000 format which may you may need special libraries to load.\n\n* sample_submission.csv A valid sample submission. Only the first few rows are available for download.\n* [train/test].csv Metadata for each patient and image. Only the first few rows of the test set are available for download.\n\n* site_id - ID code for the source hospital.\n* patient_id - ID code for the patient.\n* image_id - ID code for the image.\n* laterality - Whether the image is of the left or right breast.\n* view - The orientation of the image. The default for a screening exam is to capture two views per breast.\n* age - The patient's age in years.\n* implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n* density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult.\n* machine_id - An ID code for the imaging device.\n* cancer - The target value. Only provided for train.\n* biopsy - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n* invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n* BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n* prediction_id - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n* difficult_negative_case - True if the case was unusually difficult. Only provided for train.","metadata":{}},{"cell_type":"markdown","source":"# Target variable","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(x=train.cancer)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.564457Z","iopub.execute_input":"2023-01-10T14:16:24.56473Z","iopub.status.idle":"2023-01-10T14:16:24.770228Z","shell.execute_reply.started":"2023-01-10T14:16:24.564705Z","shell.execute_reply":"2023-01-10T14:16:24.769508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.cancer.value_counts(ascending=False))\nprint(1158/(53548+1158)*100)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.771584Z","iopub.execute_input":"2023-01-10T14:16:24.772161Z","iopub.status.idle":"2023-01-10T14:16:24.780235Z","shell.execute_reply.started":"2023-01-10T14:16:24.772123Z","shell.execute_reply":"2023-01-10T14:16:24.779305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cancer positive 2.1% only..Imabalanced dataset\n\n1 is cancer prone and 0 is negative","metadata":{}},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.781904Z","iopub.execute_input":"2023-01-10T14:16:24.782387Z","iopub.status.idle":"2023-01-10T14:16:24.789776Z","shell.execute_reply.started":"2023-01-10T14:16:24.782258Z","shell.execute_reply":"2023-01-10T14:16:24.788869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.790994Z","iopub.execute_input":"2023-01-10T14:16:24.79189Z","iopub.status.idle":"2023-01-10T14:16:24.841783Z","shell.execute_reply.started":"2023-01-10T14:16:24.791855Z","shell.execute_reply":"2023-01-10T14:16:24.840748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age ranges from 26 to 89 with a mean of 58years","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.846043Z","iopub.execute_input":"2023-01-10T14:16:24.846301Z","iopub.status.idle":"2023-01-10T14:16:24.865026Z","shell.execute_reply.started":"2023-01-10T14:16:24.846278Z","shell.execute_reply":"2023-01-10T14:16:24.864024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":">  Birads and density columns are having null values","metadata":{}},{"cell_type":"code","source":"train['density'].hist()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:24.866595Z","iopub.execute_input":"2023-01-10T14:16:24.866963Z","iopub.status.idle":"2023-01-10T14:16:25.067482Z","shell.execute_reply.started":"2023-01-10T14:16:24.866931Z","shell.execute_reply":"2023-01-10T14:16:25.066569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult.\n\nHere we are observing Least for D density!!Good but let's see other columns as well","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(x=train['age'])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:25.070475Z","iopub.execute_input":"2023-01-10T14:16:25.070753Z","iopub.status.idle":"2023-01-10T14:16:25.515784Z","shell.execute_reply.started":"2023-01-10T14:16:25.070726Z","shell.execute_reply":"2023-01-10T14:16:25.514765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['age'].hist()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:25.517422Z","iopub.execute_input":"2023-01-10T14:16:25.51786Z","iopub.status.idle":"2023-01-10T14:16:25.737847Z","shell.execute_reply.started":"2023-01-10T14:16:25.517784Z","shell.execute_reply":"2023-01-10T14:16:25.737026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:25.739375Z","iopub.execute_input":"2023-01-10T14:16:25.739716Z","iopub.status.idle":"2023-01-10T14:16:25.747969Z","shell.execute_reply.started":"2023-01-10T14:16:25.739681Z","shell.execute_reply":"2023-01-10T14:16:25.746963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.bar(train,figsize=(16,5),color='red')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:25.749494Z","iopub.execute_input":"2023-01-10T14:16:25.750201Z","iopub.status.idle":"2023-01-10T14:16:26.768962Z","shell.execute_reply.started":"2023-01-10T14:16:25.750167Z","shell.execute_reply":"2023-01-10T14:16:26.768113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['laterality'].value_counts().plot(kind='bar',figsize = (10, 5))\nplt.title('laterality')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:26.772892Z","iopub.execute_input":"2023-01-10T14:16:26.774946Z","iopub.status.idle":"2023-01-10T14:16:26.973304Z","shell.execute_reply.started":"2023-01-10T14:16:26.77491Z","shell.execute_reply":"2023-01-10T14:16:26.972436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['machine_id'].value_counts().plot(kind='bar',figsize = (10, 5),color=\"yellow\")\nplt.title('machine_id')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:26.974592Z","iopub.execute_input":"2023-01-10T14:16:26.974957Z","iopub.status.idle":"2023-01-10T14:16:27.171735Z","shell.execute_reply.started":"2023-01-10T14:16:26.974923Z","shell.execute_reply":"2023-01-10T14:16:27.170738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['difficult_negative_case'].value_counts().plot(kind='bar',figsize = (10, 5),color=\"pink\")\nplt.title('difficult_negative_case')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:27.173348Z","iopub.execute_input":"2023-01-10T14:16:27.173709Z","iopub.status.idle":"2023-01-10T14:16:27.341906Z","shell.execute_reply.started":"2023-01-10T14:16:27.173672Z","shell.execute_reply":"2023-01-10T14:16:27.341047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train[\"biopsy\"].value_counts().index\nsizes = train[\"biopsy\"].value_counts()\ncolors = ['magenta','cyan',\"orange\",\"yellow\"]\nplt.figure(figsize = (8,8))\nplt.pie(sizes, labels=labels, rotatelabels=False, autopct='%1.1f%%',colors=colors,shadow=True, startangle=90)\nplt.title('Biopsy',color = 'green',fontsize = 15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:27.343215Z","iopub.execute_input":"2023-01-10T14:16:27.34354Z","iopub.status.idle":"2023-01-10T14:16:27.480514Z","shell.execute_reply.started":"2023-01-10T14:16:27.343506Z","shell.execute_reply":"2023-01-10T14:16:27.479466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplots(figsize=(6, 6))\nsns.set_style('darkgrid')\nsns.set_palette('Pastel1')\n\ndata = [\n    train[(train[\"cancer\"] == 0) & (train[\"BIRADS\"] == 0.0)][\"image_id\"].count(),\n    train[(train[\"cancer\"] == 0) & (train[\"BIRADS\"] == 1.0)][\"image_id\"].count(),\n    train[(train[\"cancer\"] == 0) & (train[\"BIRADS\"] == 2.0)][\"image_id\"].count(),\n]\ncounts = pd.DataFrame()\n_ = plt.pie(\n    data, labels=[\"BIRADS 0\", \"BIRADS 1\", \"BIRADS 2\"],\n    autopct=lambda x: \"{:,.0f} = {:.2f}%\".format(x * sum(data)/100, x),\n    explode=[0.05] * 3, \n    pctdistance=0.5, \n    colors=sns.color_palette(\"Pastel1\")[0:3],\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:27.481916Z","iopub.execute_input":"2023-01-10T14:16:27.482505Z","iopub.status.idle":"2023-01-10T14:16:27.631985Z","shell.execute_reply.started":"2023-01-10T14:16:27.482468Z","shell.execute_reply":"2023-01-10T14:16:27.63069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are 7 main BI-RADS scores, or categories:\n* 0 - Need additional imaging evaluation\n* 1 - Negative\n* 2 - Benign\n* 3 - Probably Benign\n* 4 - Suspicious\n* 5 - Highly Suggestive of Malignancy\n* 6 - Known Biopsy-Proven Malignancy¶\n* Screening mammograms can only be assigned a BI-RADS score of 0, 1, or 2. This is why the dataset only contains 3 of the BI-RADS categories","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,8))\nsns.heatmap(train.corr(),annot=True,cmap='rainbow',fmt='.1f')","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:27.633564Z","iopub.execute_input":"2023-01-10T14:16:27.633921Z","iopub.status.idle":"2023-01-10T14:16:28.664405Z","shell.execute_reply.started":"2023-01-10T14:16:27.633887Z","shell.execute_reply":"2023-01-10T14:16:28.663498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Invasive, Biopsy are correlated with cancer in a positive manner where as BIRADS,difficult_nagative_case in oppiste manner\n\n### In this data set, the number of people for whom biopsy is requested with the suspicion of cancer is in the minority.\n\n* 🔵 In BIRADS 1-2, we see that the possibility of cancer is almost nonexistent.\n* 🔵 We see that the risk of cancer increases in the age range of 50-70 years.\n* 🔵 We see that this data set mostly consists of BI RADS-1.\n* 🔵 Since the density b and c are too much in the dataset, the cancer rate seems higher, but it may not be right to say this because the dataset contains NaN values ​​too much.\n* 🔵 A patient under 30 years of age who underwent biopsy is not possible for this data set.\n","metadata":{}},{"cell_type":"markdown","source":"# K fold","metadata":{}},{"cell_type":"code","source":"#add extra one columns\ntrain['kfold']=-1\n#Distributing the data 5 shares\nkfold = model_selection.KFold(n_splits=5, shuffle= True, random_state = 12)\nfor fold, (train_indicies, valid_indicies) in enumerate(kfold.split(X=train)):\n    #print(fold,train_indicies,valid_indicies)\n    train.loc[valid_indicies,'kfold'] = fold    \nprint(train.kfold.value_counts()) #total data 300000 = kfold split :5 * 60000\n#output of train folds data\ntrain.to_csv(\"trainfold_5.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:28.665506Z","iopub.execute_input":"2023-01-10T14:16:28.666797Z","iopub.status.idle":"2023-01-10T14:16:28.921309Z","shell.execute_reply.started":"2023-01-10T14:16:28.666757Z","shell.execute_reply":"2023-01-10T14:16:28.920302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing - Categorical to Numerical data","metadata":{}},{"cell_type":"code","source":"##Converting all categorical data to numerical\ntrain['view'] =train['view'].astype('category').cat.codes\ntrain['density'] =train['density'].astype('category').cat.codes\ntrain['laterality'] =train['laterality'].astype('category').cat.codes\ntrain['difficult_negative_case'] =train['difficult_negative_case'].astype('category').cat.codes\n\n\ntest['prediction_id'] =test['prediction_id'].astype('category').cat.codes\ntest['view'] =test['view'].astype('category').cat.codes\ntest['laterality'] = test['laterality'].astype('category').cat.codes","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:28.925805Z","iopub.execute_input":"2023-01-10T14:16:28.927906Z","iopub.status.idle":"2023-01-10T14:16:28.965376Z","shell.execute_reply.started":"2023-01-10T14:16:28.927868Z","shell.execute_reply":"2023-01-10T14:16:28.964405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:16:28.970133Z","iopub.execute_input":"2023-01-10T14:16:28.972383Z","iopub.status.idle":"2023-01-10T14:16:28.992727Z","shell.execute_reply.started":"2023-01-10T14:16:28.972346Z","shell.execute_reply":"2023-01-10T14:16:28.991892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KNN imputer","metadata":{}},{"cell_type":"markdown","source":" let us say we have variables related to the density of cars on road and levels of pollutants in the air and there are few observations that are missing for the level of pollutants, imputing the level of pollutants by mean/median level of pollutants may not necessarily be an appropriate strategy.\n\nIn such scenarios, algorithms like k-Nearest Neighbors (kNN) can help to impute the values of missing data. Sociologists and community researchers suggest that human beings live in a community because neighbors generate a feeling of security and safety, attachment to community, and relationships that bring out a community identity through participation in various activities.","metadata":{}},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\ntrain_im = pd.DataFrame(imputer.fit_transform(train))\ntest_im = pd.DataFrame(imputer.fit_transform(test))\n#remove column\ntrain_im.columns = train.columns\ntest_im.columns = test.columns\n\ntrain = train_im\ntest = test_im","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:18:58.226378Z","iopub.execute_input":"2023-01-10T14:18:58.226927Z","iopub.status.idle":"2023-01-10T14:20:23.769795Z","shell.execute_reply.started":"2023-01-10T14:18:58.226879Z","shell.execute_reply":"2023-01-10T14:20:23.768806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building the model","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor,CatBoostClassifier\nfrom xgboost import XGBRegressor,XGBClassifier\n\n#features(categorical and numerical datas separate)\nuseful_features = [c for c in train.columns if c not in (\"kfold\",\"cancer\",\"BIRADS\",\"density\",\"difficult_negative_case\",\"biopsy\")]\nobject_cols = [col for col in useful_features]\n#numerical_cols = [col for col in useful_features]\ntest = test.copy()\n\nfor fold in range(5):\n    xtrain = train[train.kfold != fold].reset_index(drop=True)\n    xvalid = train[train.kfold == fold].reset_index(drop=True)\n\n    ytrain = xtrain.cancer\n    yvalid = xvalid.cancer\n    \n    xtrain = xtrain[useful_features]\n    xvalid = xvalid[useful_features]\n        \n    #Model hyperparameter of XGboostRegressor\n    \n    xgb_params = {\n            'learning_rate': 0.001168,\n            'subsample': 0.7875490025178,\n            'colsample_bytree': 0.11807135201147,\n            'max_depth': 6,\n            'booster': 'gbtree', \n            'reg_lambda': 0.0008746338866473539,\n            'reg_alpha': 23.13181079976304,\n            'random_state':42,\n            'n_estimators':10000\n            }\n\n    model= XGBClassifier(**xgb_params)\n\n    model.fit(xtrain,ytrain,verbose=False)\n    #way of output is display\n    print(f\"fold:{fold}\")","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:17:54.684339Z","iopub.status.idle":"2023-01-10T14:17:54.686578Z","shell.execute_reply.started":"2023-01-10T14:17:54.686334Z","shell.execute_reply":"2023-01-10T14:17:54.686358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predict = model.predict(test)\ntest_predict","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:17:54.690394Z","iopub.status.idle":"2023-01-10T14:17:54.691162Z","shell.execute_reply.started":"2023-01-10T14:17:54.690921Z","shell.execute_reply":"2023-01-10T14:17:54.690945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction of data\npreds = np.mean(np.column_stack(test_predict),axis=1)\nprint(preds)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:17:54.695364Z","iopub.status.idle":"2023-01-10T14:17:54.696106Z","shell.execute_reply.started":"2023-01-10T14:17:54.695859Z","shell.execute_reply":"2023-01-10T14:17:54.695883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate submission data","metadata":{}},{"cell_type":"code","source":"sample.cancer = preds[0]\nsample.to_csv(\"submission.csv\",index=False)\nprint(\"success\")","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:17:54.697363Z","iopub.status.idle":"2023-01-10T14:17:54.698086Z","shell.execute_reply.started":"2023-01-10T14:17:54.69785Z","shell.execute_reply":"2023-01-10T14:17:54.697874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T14:17:54.705782Z","iopub.status.idle":"2023-01-10T14:17:54.706523Z","shell.execute_reply.started":"2023-01-10T14:17:54.706291Z","shell.execute_reply":"2023-01-10T14:17:54.706314Z"},"trusted":true},"execution_count":null,"outputs":[]}]}