{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# restart and clear cell output after running this","metadata":{}},{"cell_type":"code","source":"# !pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A way to reed image directly and feed it to CNN & DNN in Keras\n\n## I am new to Kaggle and the purpose of this discussion/notebook is to share a way to reed images with the associated csv data process it and feed it to a multi-input Neural Network.\n\n##  hope it will help beginners like me.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd\nimport tensorflow as tf \nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.utils import to_categorical\nfrom keras import metrics","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:19.833754Z","iopub.execute_input":"2022-12-22T14:02:19.834181Z","iopub.status.idle":"2022-12-22T14:02:23.212242Z","shell.execute_reply.started":"2022-12-22T14:02:19.834096Z","shell.execute_reply":"2022-12-22T14:02:23.210775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom as dicom\nimport cv2\nimport matplotlib.pylab as plt\nfrom PIL import Image\nfrom skimage.transform import resize\nfrom io import BytesIO","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:25.800786Z","iopub.execute_input":"2022-12-22T14:02:25.801414Z","iopub.status.idle":"2022-12-22T14:02:26.694785Z","shell.execute_reply.started":"2022-12-22T14:02:25.801384Z","shell.execute_reply":"2022-12-22T14:02:26.693643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# way to read the image and dysplay it","metadata":{}},{"cell_type":"code","source":"img_path = '../input/rsna-breast-cancer-detection/train_images/10130/388811999.dcm'\n\ndcom_img = dicom.dcmread(img_path)\nprint(dcom_img.PhotometricInterpretation)\ndata = dcom_img.pixel_array\nplt.imshow(data)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:30.063148Z","iopub.execute_input":"2022-12-22T14:02:30.063547Z","iopub.status.idle":"2022-12-22T14:02:31.894805Z","shell.execute_reply.started":"2022-12-22T14:02:30.063515Z","shell.execute_reply":"2022-12-22T14:02:31.893121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# function that return image array(np array) given the path to the image\n## RESIZE_TO is the resized size of the image the input size of the neural network should be shame as this","metadata":{}},{"cell_type":"code","source":"# RESIZE_TO  = (256, 256)\nRESIZE_TO  = (512, 512)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:38.698331Z","iopub.execute_input":"2022-12-22T14:02:38.69868Z","iopub.status.idle":"2022-12-22T14:02:38.703937Z","shell.execute_reply.started":"2022-12-22T14:02:38.698651Z","shell.execute_reply":"2022-12-22T14:02:38.702612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img_path = '../input/rsna-breast-cancer-detection/train_images/5326/1313399106.dcm'\n# img_path = '../input/rsna-breast-cancer-detection/train_images/10130/388811999.dcm'\n\ndef get_image(img_path):\n#     print(img_path)\n    dicom_img = dicom.dcmread(img_path)\n    img = dicom_img.pixel_array\n    img = (img - img.min()) / (img.max() - img.min())\n    if dicom_img.PhotometricInterpretation == \"MONOCHROME1\":  \n        img = 1 - img\n\n    image = (img * 255).astype(np.uint8)\n\n    im = Image.fromarray(image).resize(RESIZE_TO)\n\n    arr = np.asarray(im)\n    arr = np.expand_dims(arr, axis=2)\n    \n    return arr\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:41.711618Z","iopub.execute_input":"2022-12-22T14:02:41.711967Z","iopub.status.idle":"2022-12-22T14:02:41.718278Z","shell.execute_reply.started":"2022-12-22T14:02:41.71194Z","shell.execute_reply":"2022-12-22T14:02:41.717416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path = '../input/rsna-breast-cancer-detection/train_images/5326/1313399106.dcm'\narr = get_image(img_path)\nprint(arr.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:45.14314Z","iopub.execute_input":"2022-12-22T14:02:45.143471Z","iopub.status.idle":"2022-12-22T14:02:45.705667Z","shell.execute_reply.started":"2022-12-22T14:02:45.143447Z","shell.execute_reply":"2022-12-22T14:02:45.7036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# reading the train.csv to understand and prepare features data","metadata":{}},{"cell_type":"markdown","source":"# columns available in test.csv. IE.- the features that will be avvailable for prediction model.\n\n## 'site_id', 'patient_id', 'image_id', 'laterality', 'view', 'age','implant', 'machine_id'\n\n\nwe will be using age and implant and images for prediction \n","metadata":{}},{"cell_type":"code","source":"rsna_data = pd.read_csv('../input/rsna-breast-cancer-detection/train.csv')\nrsna_data.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:49.923749Z","iopub.execute_input":"2022-12-22T14:02:49.924083Z","iopub.status.idle":"2022-12-22T14:02:50.016276Z","shell.execute_reply.started":"2022-12-22T14:02:49.924054Z","shell.execute_reply":"2022-12-22T14:02:50.015619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rsna_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## there are null values in the age column. let's fill them in using the mean age","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer(strategy=\"mean\")\n\nage = pd.DataFrame(rsna_data[\"age\"])\nimputed_age = imputer.fit_transform(age)\nrsna_data['age'] = imputed_age\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:02:55.683155Z","iopub.execute_input":"2022-12-22T14:02:55.683543Z","iopub.status.idle":"2022-12-22T14:02:55.947767Z","shell.execute_reply.started":"2022-12-22T14:02:55.683514Z","shell.execute_reply":"2022-12-22T14:02:55.946269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# encode implant in one hot , but lets do that later","metadata":{}},{"cell_type":"code","source":"rsna_data['implant'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# the number of negative cases vastly overpopulate the dataset","metadata":{}},{"cell_type":"code","source":"pos_cases = (rsna_data.cancer.sum()*100)/rsna_data.size\n\nprint(\"fraction of positive cases \" , pos_cases, \"%\")\n# the dataset is vastly dominated by negetive example\n# 99.84 % accuracy if i predict all as negetive\n\nneg_cases = 100-pos_cases\nprint(\"fraction of negetive_cases / baseline prediction score \" , neg_cases, \"%\")","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:05.099885Z","iopub.execute_input":"2022-12-22T14:03:05.1003Z","iopub.status.idle":"2022-12-22T14:03:05.107623Z","shell.execute_reply.started":"2022-12-22T14:03:05.100267Z","shell.execute_reply":"2022-12-22T14:03:05.106799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_of_neg_cases = rsna_data.size - rsna_data.cancer.sum()\nmore_cases = num_of_neg_cases - rsna_data.cancer.sum()\ndrop_frac = more_cases*100/num_of_neg_cases\nprint(drop_frac)\n#drop (99 % of the neg patient data)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:08.757233Z","iopub.execute_input":"2022-12-22T14:03:08.757644Z","iopub.status.idle":"2022-12-22T14:03:08.764636Z","shell.execute_reply.started":"2022-12-22T14:03:08.757607Z","shell.execute_reply":"2022-12-22T14:03:08.763339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.__version__","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## making small dataset just for test","metadata":{}},{"cell_type":"code","source":"#drop (99 % of the neg patient data)\n# take data as multiple of 32\n\n# rsna_data_small = rsna_data.drop(rsna_data[rsna_data['cancer'] == 0].sample(frac=.90).index).reset_index(drop=True).iloc[0:6496,:]\n\nrsna_data_small = rsna_data.drop(rsna_data[rsna_data['cancer'] == 0].sample(frac=.90).index).reset_index(drop=True).iloc[0:1024,:]\n\n\nrsna_label = rsna_data_small.pop('cancer').to_numpy()\n# rsna_np = rsna_data_small.drop(['site_id', 'laterality', 'view', 'biopsy','invasive',\n#                                 'BIRADS', 'implant', 'density', 'machine_id', 'difficult_negative_case', 'img_path'], axis=1)\nrsna_np = rsna_data_small[['age', 'implant']]\nprint(rsna_np.head())\nrsna_np = rsna_np.to_numpy()\n\n\n\ndataset = tf.data.Dataset.from_tensor_slices(\n    (rsna_np.astype('float32'), rsna_label.astype('float32'), tf.strings.as_string(rsna_data_small['patient_id'].to_numpy()), \n    tf.strings.as_string(rsna_data_small['image_id'].to_numpy())))\n# img_path = '../input/rsna-breast-cancer-detection/train_images'#/5326/1313399106.dcm'\n\ndef map_fun(x, y, z1, z2):\n    return x, y, z1, z2\n\ndataset = dataset.map(map_fun)\ndataset = dataset.batch(32)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:20.142919Z","iopub.execute_input":"2022-12-22T14:03:20.143299Z","iopub.status.idle":"2022-12-22T14:03:20.295876Z","shell.execute_reply.started":"2022-12-22T14:03:20.143271Z","shell.execute_reply":"2022-12-22T14:03:20.295003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DatasetIter:\n    def __init__(self, dataset):\n        self.dataset = dataset.as_numpy_iterator()\n       \n    def __iter__(self):\n        return self\n    \n    def __next__(self):\n        try:\n            img_path = '../input/rsna-breast-cancer-detection/train_images/'\n            data = next(self.dataset)\n            patient_id = data[2].astype(str)\n            image_id = data[3].astype(str)\n            paths = []\n            for p, i in zip(patient_id, image_id):\n                paths.append(img_path+p+\"/\"+i+\".dcm\")\n                batch_length = len(paths)\n            img_tensor = np.empty((batch_length, RESIZE_TO[0], RESIZE_TO[0], 1), dtype='float')\n            i=0\n            for path in paths:\n                img_tensor[i] = get_image(path)\n                i +=1\n            i=0\n            paths = []\n\n            return (tf.convert_to_tensor(data[0]), tf.convert_to_tensor(img_tensor)), tf.convert_to_tensor(data[1])\n        except:\n            raise StopIteration\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:34.903161Z","iopub.execute_input":"2022-12-22T14:03:34.90354Z","iopub.status.idle":"2022-12-22T14:03:34.912225Z","shell.execute_reply.started":"2022-12-22T14:03:34.903512Z","shell.execute_reply":"2022-12-22T14:03:34.911246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = iter(DatasetIter(dataset))\nj = 0\nfor i in d:\n    if(j==2):\n        break\n    j+=1\n    print(i)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# loss function and metrics","metadata":{}},{"cell_type":"code","source":"def pfbeta_tf(labels, preds, beta=1):\n    preds = tf.clip_by_value(preds, 0, 1)\n    y_true_count = tf.reduce_sum(labels)\n    ctp = tf.reduce_sum(preds[labels==1])\n    cfp = tf.reduce_sum(preds[labels==0])\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0 ","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:45.405641Z","iopub.execute_input":"2022-12-22T14:03:45.406025Z","iopub.status.idle":"2022-12-22T14:03:45.416665Z","shell.execute_reply.started":"2022-12-22T14:03:45.405992Z","shell.execute_reply":"2022-12-22T14:03:45.4145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pfbeta_loss(labels, preds):\n    return -pfbeta_tf(labels, preds)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:49.703431Z","iopub.execute_input":"2022-12-22T14:03:49.703928Z","iopub.status.idle":"2022-12-22T14:03:49.708087Z","shell.execute_reply.started":"2022-12-22T14:03:49.7039Z","shell.execute_reply":"2022-12-22T14:03:49.707011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = np.array([1, 0, 1, 0])\np = np.array([1, 0, 1, 0])\n# p = np.array([0, 1, 0, 1])\nprint(pfbeta_tf(l, p))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_inputs = keras.Input(shape=(2,),)\nimage_inputs = keras.Input(shape=(RESIZE_TO[0], RESIZE_TO[0], 1))\n\nx = layers.BatchNormalization()(data_inputs)\nx = layers.Dense(8, activation=\"relu\")(x)\n\ny = layers.Rescaling(1./255)(image_inputs)\ny = layers.Conv2D(filters=32, kernel_size=3, activation=\"relu\")(y)\ny = layers.MaxPooling2D(pool_size=2)(y)\ny = layers.Conv2D(filters=64, kernel_size=3, activation=\"relu\")(y)\ny = layers.MaxPooling2D(pool_size=2)(y)\ny = layers.Conv2D(filters=128, kernel_size=3, activation=\"relu\")(y)\ny = layers.MaxPooling2D(pool_size=2)(y)\ny = layers.Conv2D(filters=256, kernel_size=3, activation=\"relu\")(y)\ny = layers.MaxPooling2D(pool_size=2)(y)\ny = layers.Conv2D(filters=256, kernel_size=3, activation=\"relu\")(y)\ny = layers.MaxPooling2D(pool_size=2)(y)\ny = layers.Conv2D(filters=256, kernel_size=3, activation=\"relu\")(y)\ny = layers.Flatten()(y)\n\nz = layers.Concatenate()([x, y]) \n\nz = layers.Dense(128, activation='relu')(z)\n\noutputs = layers.Dense(1, activation=\"sigmoid\")(z)\n\nmodel = keras.Model(inputs = [data_inputs, image_inputs], outputs=outputs)\n\nmodel.compile(optimizer=\"rmsprop\", loss=pfbeta_loss, metrics=[pfbeta_tf])\n\nkeras.utils.plot_model(model, \"rsna_model.png\", show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:03:55.669643Z","iopub.execute_input":"2022-12-22T14:03:55.669995Z","iopub.status.idle":"2022-12-22T14:03:57.179437Z","shell.execute_reply.started":"2022-12-22T14:03:55.669968Z","shell.execute_reply":"2022-12-22T14:03:57.178357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# create dataset iterator","metadata":{}},{"cell_type":"code","source":"train_dataset = iter(DatasetIter(dataset))\n# j = 0\n# for i in train_dataset:\n#     print(i)\n#     j +=1\n#     print(j)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:04:03.820485Z","iopub.execute_input":"2022-12-22T14:04:03.820864Z","iopub.status.idle":"2022-12-22T14:04:03.856558Z","shell.execute_reply.started":"2022-12-22T14:04:03.820831Z","shell.execute_reply":"2022-12-22T14:04:03.855822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_dataset,\n                    epochs=2)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:04:10.32055Z","iopub.execute_input":"2022-12-22T14:04:10.320916Z","iopub.status.idle":"2022-12-22T14:23:01.172132Z","shell.execute_reply.started":"2022-12-22T14:04:10.320887Z","shell.execute_reply":"2022-12-22T14:23:01.170159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('Rsna_1.keras')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:25:41.384438Z","iopub.execute_input":"2022-12-22T14:25:41.384933Z","iopub.status.idle":"2022-12-22T14:25:41.550367Z","shell.execute_reply.started":"2022-12-22T14:25:41.384888Z","shell.execute_reply":"2022-12-22T14:25:41.548659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = keras.models.load_model(\"../input/model/Rsna_1.keras\",\n#                                custom_objects={\"pfbeta_loss\":pfbeta_loss,\"pfbeta_tf\":pfbeta_tf, })","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prediction","metadata":{}},{"cell_type":"code","source":"img_path = '../input/rsna-breast-cancer-detection/test_images/'\ntest_dataset = pd.read_csv('../input/rsna-breast-cancer-detection/test.csv')\ntest_dataset['img_path'] = img_path+test_dataset['patient_id'].astype('str') + \"/\" + test_dataset['image_id'].astype('str') +\".dcm\"\ntest_dataset.head().to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:25:44.60041Z","iopub.execute_input":"2022-12-22T14:25:44.600774Z","iopub.status.idle":"2022-12-22T14:25:44.627097Z","shell.execute_reply.started":"2022-12-22T14:25:44.600746Z","shell.execute_reply":"2022-12-22T14:25:44.625889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.columns","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:25:48.127617Z","iopub.execute_input":"2022-12-22T14:25:48.127999Z","iopub.status.idle":"2022-12-22T14:25:48.136668Z","shell.execute_reply.started":"2022-12-22T14:25:48.127969Z","shell.execute_reply":"2022-12-22T14:25:48.135433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:25:50.96385Z","iopub.execute_input":"2022-12-22T14:25:50.964281Z","iopub.status.idle":"2022-12-22T14:25:50.978976Z","shell.execute_reply.started":"2022-12-22T14:25:50.964248Z","shell.execute_reply":"2022-12-22T14:25:50.977845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_small = test_dataset.drop(['site_id', 'patient_id','image_id', 'laterality', 'view', 'machine_id', 'prediction_id', 'img_path'], axis=1)\ntest_small.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:25:54.853043Z","iopub.execute_input":"2022-12-22T14:25:54.853464Z","iopub.status.idle":"2022-12-22T14:25:54.866154Z","shell.execute_reply.started":"2022-12-22T14:25:54.853432Z","shell.execute_reply":"2022-12-22T14:25:54.865014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path = '../input/rsna-breast-cancer-detection/test_images/'","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:25:59.365334Z","iopub.execute_input":"2022-12-22T14:25:59.366507Z","iopub.status.idle":"2022-12-22T14:25:59.37076Z","shell.execute_reply.started":"2022-12-22T14:25:59.366457Z","shell.execute_reply":"2022-12-22T14:25:59.369811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = np.empty((4, 2), dtype='float')\ntest_images = np.empty((4, RESIZE_TO[0], RESIZE_TO[0], 1), dtype='float')\ni=0\nfeatures = test_small.to_numpy()\nimg_paths = test_dataset['img_path'].to_numpy()\nfor i in range(len(features)):\n    test_data[i] = features[i]\n#     print(get_image(img_paths[i]).shape)\n    test_images[i] = get_image(img_paths[i])\n    \n# test_data.append((tf.convert_to_tensor(features[i])))\n\nprint(test_data.shape)\nprint(test_images.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:26:13.871014Z","iopub.execute_input":"2022-12-22T14:26:13.871487Z","iopub.status.idle":"2022-12-22T14:26:16.127963Z","shell.execute_reply.started":"2022-12-22T14:26:13.871455Z","shell.execute_reply":"2022-12-22T14:26:16.12677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_pred = model.predict([test_data,test_images])\nprint(cancer_pred)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:26:21.36734Z","iopub.execute_input":"2022-12-22T14:26:21.367711Z","iopub.status.idle":"2022-12-22T14:26:22.060064Z","shell.execute_reply.started":"2022-12-22T14:26:21.367683Z","shell.execute_reply":"2022-12-22T14:26:22.059087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = test_dataset['prediction_id']\ncancer_pred = np.squeeze(cancer_pred).tolist()\ncancer_pred = pd.Series(cancer_pred, name=\"cancer\")\npredictions =  pd.concat([predictions, cancer_pred], axis=1)\nprint(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:26:27.793881Z","iopub.execute_input":"2022-12-22T14:26:27.794257Z","iopub.status.idle":"2022-12-22T14:26:27.803999Z","shell.execute_reply.started":"2022-12-22T14:26:27.794229Z","shell.execute_reply":"2022-12-22T14:26:27.802527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = predictions.drop_duplicates(subset='prediction_id')\nprint(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:26:32.576932Z","iopub.execute_input":"2022-12-22T14:26:32.577308Z","iopub.status.idle":"2022-12-22T14:26:32.588386Z","shell.execute_reply.started":"2022-12-22T14:26:32.57728Z","shell.execute_reply":"2022-12-22T14:26:32.587236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = predictions\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T14:26:36.447169Z","iopub.execute_input":"2022-12-22T14:26:36.447524Z","iopub.status.idle":"2022-12-22T14:26:36.457176Z","shell.execute_reply.started":"2022-12-22T14:26:36.447499Z","shell.execute_reply":"2022-12-22T14:26:36.456242Z"},"trusted":true},"execution_count":null,"outputs":[]}]}