{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport pydicom\nfrom keras import layers\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import Callback, ModelCheckpoint, EarlyStopping\nfrom keras.initializers import Constant\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.python.ops import array_ops\nfrom tqdm import tqdm\nfrom keras import backend as K\nimport tensorflow as tf\nimport keras\nfrom keras.applications import Xception\nfrom keras.models import Model, load_model\nfrom math import ceil, floor\nfrom sklearn.model_selection import ShuffleSplit\nfrom sklearn.metrics import log_loss\nfrom keras.layers import Dense, Flatten, Dropout, GlobalAveragePooling2D","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-26T13:37:44.058808Z","iopub.execute_input":"2021-08-26T13:37:44.059203Z","iopub.status.idle":"2021-08-26T13:37:50.235062Z","shell.execute_reply.started":"2021-08-26T13:37:44.059116Z","shell.execute_reply":"2021-08-26T13:37:50.23382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*** Things to look at for optmization: **\n* Different file sizes\n* Different Batch sizes and more/less epoch \n* using different learning rate or metrics (AUC) \n* Sensitivy and specificty \n* Only take 'any\" or most frequent -> make tables \n* Get balanced data \n* Batch normalization \n* adaptive loss function\n* Data augmentation? GANs?","metadata":{}},{"cell_type":"markdown","source":"**Testing/prediction**\n* Fill the sample documents provided in the dataset + test on validation data\n* Try to make a scrpit that takes an image, and return the probability for each label \n* Makes a labeled validation set? \n* Other ways to validate ","metadata":{}},{"cell_type":"code","source":"os.listdir('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection')","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:38:13.321541Z","iopub.execute_input":"2021-08-26T13:38:13.321959Z","iopub.status.idle":"2021-08-26T13:38:13.344377Z","shell.execute_reply.started":"2021-08-26T13:38:13.321928Z","shell.execute_reply":"2021-08-26T13:38:13.343642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/'\nTRAIN_DIR = 'stage_2_train/'\nTEST_DIR = 'stage_2_test/'\ntrain_df = pd.read_csv(BASE_PATH + 'stage_2_train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:38:23.90419Z","iopub.execute_input":"2021-08-26T13:38:23.904524Z","iopub.status.idle":"2021-08-26T13:38:27.61912Z","shell.execute_reply.started":"2021-08-26T13:38:23.904495Z","shell.execute_reply":"2021-08-26T13:38:27.618209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv(BASE_PATH + 'stage_2_sample_submission.csv')\n\ntrain_df['filename'] = train_df['ID'].apply(lambda st: \"ID_\" + st.split('_')[1] + \".png\")\ntrain_df['type'] = train_df['ID'].apply(lambda st: st.split('_')[2])\nsub_df['filename'] = sub_df['ID'].apply(lambda st: \"ID_\" + st.split('_')[1] + \".png\")\nsub_df['type'] = sub_df['ID'].apply(lambda st: st.split('_')[2])\n\nprint(train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:38:27.620528Z","iopub.execute_input":"2021-08-26T13:38:27.620878Z","iopub.status.idle":"2021-08-26T13:38:34.526455Z","shell.execute_reply.started":"2021-08-26T13:38:27.620827Z","shell.execute_reply":"2021-08-26T13:38:34.525329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(sub_df.filename.unique(), columns=['filename'])\nprint(test_df.shape)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:40:15.199908Z","iopub.execute_input":"2021-08-26T13:40:15.200265Z","iopub.status.idle":"2021-08-26T13:40:15.290934Z","shell.execute_reply.started":"2021-08-26T13:40:15.200233Z","shell.execute_reply":"2021-08-26T13:40:15.289973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_test_df = pd.DataFrame(sub_df.filename.unique(), columns=['filename'])\nprint(png_test_df.shape)\npng_test_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:40:16.923568Z","iopub.execute_input":"2021-08-26T13:40:16.923881Z","iopub.status.idle":"2021-08-26T13:40:17.013687Z","shell.execute_reply.started":"2021-08-26T13:40:16.923853Z","shell.execute_reply":"2021-08-26T13:40:17.012736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcm_df = test_df\ndcm_df['filename'] = dcm_df['filename'].apply(lambda x: x.replace('.png', '.dcm'))\ndcm_df","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:40:19.455787Z","iopub.execute_input":"2021-08-26T13:40:19.456177Z","iopub.status.idle":"2021-08-26T13:40:19.524249Z","shell.execute_reply.started":"2021-08-26T13:40:19.456137Z","shell.execute_reply":"2021-08-26T13:40:19.523305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_test_df","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:40:21.286249Z","iopub.execute_input":"2021-08-26T13:40:21.287374Z","iopub.status.idle":"2021-08-26T13:40:21.312637Z","shell.execute_reply.started":"2021-08-26T13:40:21.286971Z","shell.execute_reply":"2021-08-26T13:40:21.311684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subtypes = train_df.groupby('type').sum()\nsubtypes","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:36:17.853908Z","iopub.execute_input":"2021-08-25T15:36:17.854236Z","iopub.status.idle":"2021-08-25T15:36:19.061631Z","shell.execute_reply.started":"2021-08-25T15:36:17.854206Z","shell.execute_reply":"2021-08-25T15:36:19.060852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(y=subtypes.index, x=subtypes.Label, palette=\"deep\")","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:36:21.681821Z","iopub.execute_input":"2021-08-25T15:36:21.682164Z","iopub.status.idle":"2021-08-25T15:36:22.03717Z","shell.execute_reply.started":"2021-08-25T15:36:21.682134Z","shell.execute_reply":"2021-08-25T15:36:22.036298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**From this we can note a few things: **\n* There are 107933 images with \"any\" hemmorhages. This is quite low compared to the 720000 images we have in the dataset \n* Thus, we could create a generator with all the images containing hemorrages and the same amount of images not subject to an hemmorhage. \n* Additionaly, all types of hemmorhages are realtively equaly represented, except the 'epidural' type that has only 3145 cases in the whole dataset. We could try to run in with this type discarded.  ","metadata":{}},{"cell_type":"code","source":"np.random.seed(2019)\nsample_files = np.random.choice(os.listdir(BASE_PATH + TRAIN_DIR), 400000) # take the rest for testing\nsample_df = train_df[train_df.filename.apply(lambda x: x.replace('.png', '.dcm')).isin(sample_files)]","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:36:23.791877Z","iopub.execute_input":"2021-08-25T15:36:23.792228Z","iopub.status.idle":"2021-08-25T15:36:53.628107Z","shell.execute_reply.started":"2021-08-25T15:36:23.792197Z","shell.execute_reply":"2021-08-25T15:36:53.627184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pivot_df = sample_df[['Label', 'filename', 'type']].drop_duplicates().pivot(\n    index='filename', columns='type', values='Label').reset_index()\nprint(pivot_df.shape)\npivot_df","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:36:53.629635Z","iopub.execute_input":"2021-08-25T15:36:53.629975Z","iopub.status.idle":"2021-08-25T15:36:53.80147Z","shell.execute_reply.started":"2021-08-25T15:36:53.629941Z","shell.execute_reply":"2021-08-25T15:36:53.800511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#one_df = pivot_df.drop(pivot_df.loc[pivot_df['subdural']==0].index)\n#one_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:44.365987Z","iopub.execute_input":"2021-07-22T14:22:44.366352Z","iopub.status.idle":"2021-07-22T14:22:44.370581Z","shell.execute_reply.started":"2021-07-22T14:22:44.366313Z","shell.execute_reply":"2021-07-22T14:22:44.369442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#zero_df = pivot_df.drop(pivot_df.loc[pivot_df['any']==1].index)\n#zero_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:44.371953Z","iopub.execute_input":"2021-07-22T14:22:44.372327Z","iopub.status.idle":"2021-07-22T14:22:44.379479Z","shell.execute_reply.started":"2021-07-22T14:22:44.372291Z","shell.execute_reply":"2021-07-22T14:22:44.378563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#zero_df = zero_df.sample(47166)\n#zero_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:44.380766Z","iopub.execute_input":"2021-07-22T14:22:44.381134Z","iopub.status.idle":"2021-07-22T14:22:44.387208Z","shell.execute_reply.started":"2021-07-22T14:22:44.381098Z","shell.execute_reply":"2021-07-22T14:22:44.386358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample_df = pd.concat([zero_df, one_df])\n#sample_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:44.388304Z","iopub.execute_input":"2021-07-22T14:22:44.388656Z","iopub.status.idle":"2021-07-22T14:22:44.395716Z","shell.execute_reply.started":"2021-07-22T14:22:44.388611Z","shell.execute_reply":"2021-07-22T14:22:44.394914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#zero_df = pivot_df.drop(pivot_df.loc[pivot_df['any']==1].index)\n#zero_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:51.990217Z","iopub.execute_input":"2021-07-22T14:22:51.990716Z","iopub.status.idle":"2021-07-22T14:22:51.999408Z","shell.execute_reply.started":"2021-07-22T14:22:51.99068Z","shell.execute_reply":"2021-07-22T14:22:51.998137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#zero_df = zero_df.sample(47166)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:52.000704Z","iopub.execute_input":"2021-07-22T14:22:52.001235Z","iopub.status.idle":"2021-07-22T14:22:52.009036Z","shell.execute_reply.started":"2021-07-22T14:22:52.001182Z","shell.execute_reply":"2021-07-22T14:22:52.008234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample_df = pd.concat([zero_df, one_df])\n#sample_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:52.009923Z","iopub.execute_input":"2021-07-22T14:22:52.01027Z","iopub.status.idle":"2021-07-22T14:22:52.019217Z","shell.execute_reply.started":"2021-07-22T14:22:52.010235Z","shell.execute_reply":"2021-07-22T14:22:52.018399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.utils import shuffle\n#sample_df = shuffle(sample_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:52.020209Z","iopub.execute_input":"2021-07-22T14:22:52.020525Z","iopub.status.idle":"2021-07-22T14:22:52.029043Z","shell.execute_reply.started":"2021-07-22T14:22:52.020494Z","shell.execute_reply":"2021-07-22T14:22:52.028236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_df = pivot_df.sample(int(len(pivot_df) * 0.15))  \nvalidation_df ","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:36:57.295342Z","iopub.execute_input":"2021-08-25T15:36:57.295748Z","iopub.status.idle":"2021-08-25T15:36:57.316322Z","shell.execute_reply.started":"2021-08-25T15:36:57.295714Z","shell.execute_reply":"2021-08-25T15:36:57.315515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = []\nfor i in range(len(validation_df)): \n    y_true.append(validation_df.iloc[i,1])\n        \n","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:36:59.717438Z","iopub.execute_input":"2021-08-25T15:36:59.717767Z","iopub.status.idle":"2021-08-25T15:36:59.795961Z","shell.execute_reply.started":"2021-08-25T15:36:59.717739Z","shell.execute_reply":"2021-08-25T15:36:59.795206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y_true)","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:01.83596Z","iopub.execute_input":"2021-08-25T15:37:01.836275Z","iopub.status.idle":"2021-08-25T15:37:01.843606Z","shell.execute_reply.started":"2021-08-25T15:37:01.836247Z","shell.execute_reply":"2021-08-25T15:37:01.842303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_true = []\nfor i in range(len(validation_df)): \n    for j in range(1,7): \n        full_true.append(validation_df.iloc[i,j])\n        \n","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:03.717987Z","iopub.execute_input":"2021-08-25T15:37:03.71834Z","iopub.status.idle":"2021-08-25T15:37:04.143044Z","shell.execute_reply.started":"2021-08-25T15:37:03.718303Z","shell.execute_reply":"2021-08-25T15:37:04.142221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(full_true)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:22:52.968486Z","iopub.execute_input":"2021-07-22T14:22:52.968882Z","iopub.status.idle":"2021-07-22T14:22:52.972306Z","shell.execute_reply.started":"2021-07-22T14:22:52.968845Z","shell.execute_reply":"2021-07-22T14:22:52.971523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_df = pivot_df[~(pivot_df.filename.isin(validation_df.filename))]\ntraining_df\n","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:07.020605Z","iopub.execute_input":"2021-08-25T15:37:07.020946Z","iopub.status.idle":"2021-08-25T15:37:07.045203Z","shell.execute_reply.started":"2021-08-25T15:37:07.020918Z","shell.execute_reply":"2021-08-25T15:37:07.044257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(training_df.head())\nprint(validation_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:08.90712Z","iopub.execute_input":"2021-08-25T15:37:08.907459Z","iopub.status.idle":"2021-08-25T15:37:08.920731Z","shell.execute_reply.started":"2021-08-25T15:37:08.907427Z","shell.execute_reply":"2021-08-25T15:37:08.91989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pixels_hu(scan): \n    image = np.stack([scan.pixel_array])\n    image = image.astype(np.int16) \n    \n    image[image == -2000] = 0\n    \n    intercept = scan.RescaleIntercept\n    slope = scan.RescaleSlope\n    \n    if slope != 1: \n        image = slope * image.astype(np.float64)\n        image = image.astype(np.int16)\n    \n    image += np.int16(intercept) \n    \n    return np.array(image, dtype=np.int16)","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:10.497239Z","iopub.execute_input":"2021-08-25T15:37:10.497614Z","iopub.status.idle":"2021-08-25T15:37:10.504263Z","shell.execute_reply.started":"2021-08-25T15:37:10.497582Z","shell.execute_reply":"2021-08-25T15:37:10.503077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_window(image, center, width):\n    image = image.copy()\n    min_value = center - width // 2\n    max_value = center + width // 2\n    image[image < min_value] = min_value\n    image[image > max_value] = max_value\n    return image\n\n\ndef apply_window_policy(image):\n\n    image1 = apply_window(image, 40, 80) # brain\n    image2 = apply_window(image, 80, 200) # subdural\n    image3 = apply_window(image, 40, 380) # bone\n    image1 = (image1 - 0) / 80\n    image2 = (image2 - (-20)) / 200\n    image3 = (image3 - (-150)) / 380\n    image = np.array([\n        image1 - image1.mean(),\n        image2 - image2.mean(),\n        image3 - image3.mean(),\n    ]).transpose(1,2,0)\n\n    return image\n#maybe try a new function ","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:11.98657Z","iopub.execute_input":"2021-08-25T15:37:11.986899Z","iopub.status.idle":"2021-08-25T15:37:11.993527Z","shell.execute_reply.started":"2021-08-25T15:37:11.986863Z","shell.execute_reply":"2021-08-25T15:37:11.992685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_and_resize(filenames, load_dir):    \n    save_dir = '/kaggle/tmp/'\n    if not os.path.exists(save_dir):\n        os.makedirs(save_dir)\n\n    for filename in tqdm(filenames):\n        try:\n            path = load_dir + filename\n            new_path = save_dir + filename.replace('.dcm', '.png')\n            dcm = pydicom.dcmread(path)\n            image = get_pixels_hu(dcm)\n            image = apply_window_policy(image[0])\n            image -= image.min((0,1))\n            image = (255*image).astype(np.uint8)\n            image = cv2.resize(image, (299, 299)) #smaller\n            res = cv2.imwrite(new_path, image)\n            \n        except ValueError:\n            continue # it returns a black image, super weird ","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:13.472286Z","iopub.execute_input":"2021-08-25T15:37:13.472658Z","iopub.status.idle":"2021-08-25T15:37:13.478969Z","shell.execute_reply.started":"2021-08-25T15:37:13.472627Z","shell.execute_reply":"2021-08-25T15:37:13.478009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_and_resize(filenames=sample_files, load_dir=BASE_PATH + TRAIN_DIR)\nsave_and_resize(filenames=dcm_df.filename, load_dir=BASE_PATH + TEST_DIR)","metadata":{"execution":{"iopub.status.busy":"2021-08-25T15:37:15.396395Z","iopub.execute_input":"2021-08-25T15:37:15.396725Z","iopub.status.idle":"2021-08-25T16:52:39.790598Z","shell.execute_reply.started":"2021-08-25T15:37:15.396697Z","shell.execute_reply":"2021-08-25T16:52:39.789301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():    \n    base_model = Xception(weights = 'imagenet', include_top = False, input_shape = (299,299,3))\n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(0.15)(x)\n    y_pred = Dense(6, activation = 'sigmoid')(x)\n\n    return Model(inputs = base_model.input, outputs = y_pred)","metadata":{"execution":{"iopub.status.busy":"2021-08-25T16:53:07.738349Z","iopub.execute_input":"2021-08-25T16:53:07.738717Z","iopub.status.idle":"2021-08-25T16:53:07.746531Z","shell.execute_reply.started":"2021-08-25T16:53:07.738685Z","shell.execute_reply":"2021-08-25T16:53:07.745695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LR = 0.00005\nmodel = create_model()","metadata":{"execution":{"iopub.status.busy":"2021-08-25T16:53:09.06341Z","iopub.execute_input":"2021-08-25T16:53:09.063752Z","iopub.status.idle":"2021-08-25T16:53:12.830644Z","shell.execute_reply.started":"2021-08-25T16:53:09.063721Z","shell.execute_reply":"2021-08-25T16:53:12.82977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from keras.utils.vis_utils import plot_model\n#plot_model(model, to_file='model_plot.png', show_shapes=True, show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:42:29.556603Z","iopub.execute_input":"2021-07-22T14:42:29.557Z","iopub.status.idle":"2021-07-22T14:42:29.562424Z","shell.execute_reply.started":"2021-07-22T14:42:29.556961Z","shell.execute_reply":"2021-07-22T14:42:29.560745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer = Adam(learning_rate = LR), \n              loss = 'binary_crossentropy', # <- requires balance/ Binary for unbalanced\n              metrics = [tf.keras.metrics.SensitivityAtSpecificity(0.5)]) #run both ","metadata":{"execution":{"iopub.status.busy":"2021-08-25T16:53:12.832122Z","iopub.execute_input":"2021-08-25T16:53:12.832468Z","iopub.status.idle":"2021-08-25T16:53:12.859657Z","shell.execute_reply.started":"2021-08-25T16:53:12.832431Z","shell.execute_reply":"2021-08-25T16:53:12.85862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras_preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2021-08-25T16:53:14.016229Z","iopub.execute_input":"2021-08-25T16:53:14.016589Z","iopub.status.idle":"2021-08-25T16:53:14.020701Z","shell.execute_reply.started":"2021-08-25T16:53:14.016556Z","shell.execute_reply":"2021-08-25T16:53:14.019465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2021-08-25T16:53:56.145822Z","iopub.execute_input":"2021-08-25T16:53:56.146153Z","iopub.status.idle":"2021-08-25T16:53:56.157925Z","shell.execute_reply.started":"2021-08-25T16:53:56.146123Z","shell.execute_reply":"2021-08-25T16:53:56.156881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16 # had to revert back to 16 to have a comparaison point with the large model I ran locally \n\ndef create_datagen():\n    return ImageDataGenerator()\n\ndef create_test_gen():\n    return ImageDataGenerator().flow_from_dataframe(\n        png_test_df,\n        directory=  '/kaggle/tmp/',\n        x_col='filename',\n        class_mode=None,\n        target_size=(299, 299),\n        batch_size=BATCH_SIZE,\n        shuffle=False\n    )\n\ndef create_train_gen(datagen):\n    return datagen.flow_from_dataframe(\n        training_df, \n        directory='/kaggle/tmp/',\n        \n        x_col='filename', \n        y_col=['any', 'epidural', 'intraparenchymal', \n               'intraventricular', 'subarachnoid', 'subdural'],\n        class_mode='raw',\n        target_size=(299, 299),\n        batch_size=BATCH_SIZE,\n        \n       \n    )\ndef create_val_gen(datagen): \n    return datagen.flow_from_dataframe(\n        validation_df, \n        directory='/kaggle/tmp/',\n        \n        x_col='filename', \n        y_col=['any', 'epidural', 'intraparenchymal', \n               'intraventricular', 'subarachnoid', 'subdural'],\n        class_mode='raw',\n        target_size=(299, 299),\n        batch_size=BATCH_SIZE,\n        shuffle=False,\n        \n    )\n\n# Using original generator\ndata_generator = create_datagen()\ntrain_gen = create_train_gen(data_generator)\nval_gen = create_val_gen(data_generator)\ntest_gen = create_test_gen()","metadata":{"execution":{"iopub.status.busy":"2021-08-25T16:55:13.451418Z","iopub.execute_input":"2021-08-25T16:55:13.45175Z","iopub.status.idle":"2021-08-25T16:55:15.313912Z","shell.execute_reply.started":"2021-08-25T16:55:13.45172Z","shell.execute_reply":"2021-08-25T16:55:15.312727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:42:30.082475Z","iopub.execute_input":"2021-07-22T14:42:30.083028Z","iopub.status.idle":"2021-07-22T14:42:30.153Z","shell.execute_reply.started":"2021-07-22T14:42:30.082982Z","shell.execute_reply":"2021-07-22T14:42:30.152175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    'effnetb4.h5', \n    monitor='val_loss', \n    verbose=0, \n    save_best_only=True, \n    save_weights_only=False,\n    mode='auto'\n)\nEarly_stop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', min_delta=0, patience=0, verbose=1, \n                                              mode='auto', baseline=None, restore_best_weights=False)\n#train_length = len(train_df)\ntotal_steps = sample_files.shape[0] // BATCH_SIZE\ntotal_steps = total_steps // 4\nhistory = model.fit_generator(\n    train_gen,\n    steps_per_epoch = total_steps,\n    validation_data=val_gen,\n    validation_steps=total_steps * 0.15,\n    callbacks=[checkpoint, Early_stop],\n    epochs=10\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:17:23.120095Z","iopub.execute_input":"2021-07-23T17:17:23.120452Z","iopub.status.idle":"2021-07-23T17:30:04.979849Z","shell.execute_reply.started":"2021-07-23T17:17:23.12042Z","shell.execute_reply":"2021-07-23T17:30:04.979009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['sensitivity_at_specificity']\nval_acc = history.history['val_sensitivity_at_specificity']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, acc, 'b', label='Training Sens')\nplt.plot(epochs, val_acc, 'g', label='Validation Sens')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\n\nplt.title('Training and validation accuracy')\nplt.legend()\nfig = plt.figure()\nfig.savefig('acc.png')\n\n\nplt.plot(epochs, loss, 'b', label='Training loss')\nplt.plot(epochs, val_loss, 'g', label='Validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Training and validation loss')\n\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:30:04.981724Z","iopub.execute_input":"2021-07-23T17:30:04.982074Z","iopub.status.idle":"2021-07-23T17:30:05.289969Z","shell.execute_reply.started":"2021-07-23T17:30:04.982037Z","shell.execute_reply":"2021-07-23T17:30:05.288423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Idea here: make pred on validation, then for each image load the image, the prediciton, and the labels in validation_df ","metadata":{}},{"cell_type":"code","source":"test_preds = model.predict_generator(test_gen, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:32:23.025842Z","iopub.execute_input":"2021-07-23T17:32:23.026201Z","iopub.status.idle":"2021-07-23T17:41:54.221234Z","shell.execute_reply.started":"2021-07-23T17:32:23.02617Z","shell.execute_reply":"2021-07-23T17:41:54.220275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:51:03.929757Z","iopub.execute_input":"2021-07-23T17:51:03.930093Z","iopub.status.idle":"2021-07-23T17:51:03.936944Z","shell.execute_reply.started":"2021-07-23T17:51:03.930065Z","shell.execute_reply":"2021-07-23T17:51:03.935781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.join(pd.DataFrame(test_preds, columns=[\n    'any', 'epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural'\n]))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:53:33.462333Z","iopub.execute_input":"2021-07-23T17:53:33.462772Z","iopub.status.idle":"2021-07-23T17:53:33.501678Z","shell.execute_reply.started":"2021-07-23T17:53:33.462739Z","shell.execute_reply":"2021-07-23T17:53:33.499748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.melt(id_vars=['filename'])\n\n# Combine the filename column with the variable column\ntest_df['ID'] = test_df.filename.apply(lambda x: x.replace('.png', '')) + '_' + test_df.variable\ntest_df['Label'] = test_df['value']","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:53:58.15212Z","iopub.execute_input":"2021-07-23T17:53:58.152488Z","iopub.status.idle":"2021-07-23T17:53:58.776119Z","shell.execute_reply.started":"2021-07-23T17:53:58.152446Z","shell.execute_reply":"2021-07-23T17:53:58.775125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[['ID', 'Label']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:54:17.931492Z","iopub.execute_input":"2021-07-23T17:54:17.931871Z","iopub.status.idle":"2021-07-23T17:54:20.961922Z","shell.execute_reply.started":"2021-07-23T17:54:17.931839Z","shell.execute_reply":"2021-07-23T17:54:20.961054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_preds = model.predict_generator(val_gen, verbose = 1)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:05.086689Z","iopub.execute_input":"2021-07-22T14:53:05.087081Z","iopub.status.idle":"2021-07-22T14:53:19.859619Z","shell.execute_reply.started":"2021-07-22T14:53:05.087041Z","shell.execute_reply":"2021-07-22T14:53:19.858829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_preds","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:19.861106Z","iopub.execute_input":"2021-07-22T14:53:19.861473Z","iopub.status.idle":"2021-07-22T14:53:19.87705Z","shell.execute_reply.started":"2021-07-22T14:53:19.861433Z","shell.execute_reply":"2021-07-22T14:53:19.872679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_preds = []\n#for i in range(len(val_preds)):\n#    y_preds.append(0)\n#    for value in val_preds[i]: \n#        if value > 0.5: \n#            y_preds[i] = 1\n#            break\n            \n        \n#len(y_preds)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:19.879243Z","iopub.execute_input":"2021-07-22T14:53:19.879752Z","iopub.status.idle":"2021-07-22T14:53:19.977685Z","shell.execute_reply.started":"2021-07-22T14:53:19.8797Z","shell.execute_reply":"2021-07-22T14:53:19.976316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.metrics import roc_curve\n#fpr_keras, tpr_keras, thresholds_keras = roc_curve(y_true, y_preds)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:19.979539Z","iopub.execute_input":"2021-07-22T14:53:19.980317Z","iopub.status.idle":"2021-07-22T14:53:19.992892Z","shell.execute_reply.started":"2021-07-22T14:53:19.980278Z","shell.execute_reply":"2021-07-22T14:53:19.991829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.metrics import auc\n#auc_keras = auc(fpr_keras, tpr_keras)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:19.994244Z","iopub.execute_input":"2021-07-22T14:53:19.996107Z","iopub.status.idle":"2021-07-22T14:53:20.003506Z","shell.execute_reply.started":"2021-07-22T14:53:19.996058Z","shell.execute_reply":"2021-07-22T14:53:20.002667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(1)\n#plt.plot([0, 1], [0, 1], 'k--')\n#plt.plot(fpr_keras, tpr_keras, label='Keras (area = {:.3f})'.format(auc_keras))\n\n#plt.xlabel('False positive rate')\n#plt.ylabel('True positive rate')\n#plt.title('ROC curve')\n#plt.legend(loc='best')\n#plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.004929Z","iopub.execute_input":"2021-07-22T14:53:20.005429Z","iopub.status.idle":"2021-07-22T14:53:20.252426Z","shell.execute_reply.started":"2021-07-22T14:53:20.005391Z","shell.execute_reply":"2021-07-22T14:53:20.251651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.metrics import confusion_matrix\n#print('2*2 Confusion Matrix')\n#print(confusion_matrix(y_true, y_preds))\n#cm = confusion_matrix(y_true, y_preds)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.256044Z","iopub.execute_input":"2021-07-22T14:53:20.25798Z","iopub.status.idle":"2021-07-22T14:53:20.290566Z","shell.execute_reply.started":"2021-07-22T14:53:20.257935Z","shell.execute_reply":"2021-07-22T14:53:20.28982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools   \ndef plot_confusion_matrix(cm, classes,\n                        normalize=False,\n                        title='Confusion matrix',\n                        cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n            horizontalalignment=\"center\",\n            color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.294184Z","iopub.execute_input":"2021-07-22T14:53:20.296213Z","iopub.status.idle":"2021-07-22T14:53:20.309161Z","shell.execute_reply.started":"2021-07-22T14:53:20.296172Z","shell.execute_reply":"2021-07-22T14:53:20.30822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_labels = ['no hemorrhage', 'has hemorrhage']","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.312886Z","iopub.execute_input":"2021-07-22T14:53:20.315244Z","iopub.status.idle":"2021-07-22T14:53:20.320482Z","shell.execute_reply.started":"2021-07-22T14:53:20.315205Z","shell.execute_reply":"2021-07-22T14:53:20.319678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot_confusion_matrix(cm=cm, classes=cm_labels, title='Confusion Matrix')","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.324294Z","iopub.execute_input":"2021-07-22T14:53:20.326356Z","iopub.status.idle":"2021-07-22T14:53:20.587662Z","shell.execute_reply.started":"2021-07-22T14:53:20.326316Z","shell.execute_reply":"2021-07-22T14:53:20.586932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions_list = []\n#for pred in val_preds: \n  #  predictions_list.append(pred)\n\nlen(predictions_list)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.590382Z","iopub.execute_input":"2021-07-22T14:53:20.590626Z","iopub.status.idle":"2021-07-22T14:53:20.599649Z","shell.execute_reply.started":"2021-07-22T14:53:20.590596Z","shell.execute_reply":"2021-07-22T14:53:20.598683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_frame = validation_df.drop(['filename'], axis=1)\nvalidation_frame ","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.601178Z","iopub.execute_input":"2021-07-22T14:53:20.601591Z","iopub.status.idle":"2021-07-22T14:53:20.616444Z","shell.execute_reply.started":"2021-07-22T14:53:20.601556Z","shell.execute_reply":"2021-07-22T14:53:20.615244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(validation_frame) ","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.617944Z","iopub.execute_input":"2021-07-22T14:53:20.61844Z","iopub.status.idle":"2021-07-22T14:53:20.625089Z","shell.execute_reply.started":"2021-07-22T14:53:20.618402Z","shell.execute_reply":"2021-07-22T14:53:20.62398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(predictions_list) == len(validation_frame): \n    validation_frame.iloc[:,:] = predictions_list\nelse: \n    print(\"fix this issue\")\n        ","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.626656Z","iopub.execute_input":"2021-07-22T14:53:20.627077Z","iopub.status.idle":"2021-07-22T14:53:20.645511Z","shell.execute_reply.started":"2021-07-22T14:53:20.627042Z","shell.execute_reply":"2021-07-22T14:53:20.64481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_frame.insert(0, \"filename\", validation_df.filename)\nvalidation_frame.insert(7, \"true_any\" ,validation_df.iloc[:,1])\nvalidation_frame.insert(8, \"true_epidural\", validation_df.epidural)\nvalidation_frame.insert(9, \"true_intraparenchymal\", validation_df.intraparenchymal)\nvalidation_frame.insert(10, \"true_intraventricular\", validation_df.intraventricular)\nvalidation_frame.insert(11, \"true_subarachnoid\", validation_df.subarachnoid)\nvalidation_frame.insert(12, \"true_subdural\", validation_df.subdural)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.646624Z","iopub.execute_input":"2021-07-22T14:53:20.646965Z","iopub.status.idle":"2021-07-22T14:53:20.660804Z","shell.execute_reply.started":"2021-07-22T14:53:20.646932Z","shell.execute_reply":"2021-07-22T14:53:20.659755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_frame","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.662271Z","iopub.execute_input":"2021-07-22T14:53:20.662655Z","iopub.status.idle":"2021-07-22T14:53:20.688196Z","shell.execute_reply.started":"2021-07-22T14:53:20.662579Z","shell.execute_reply":"2021-07-22T14:53:20.687379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(100): \n    if validation_frame.iloc[i,1] > 0.8: \n        print(\"ID is : \" + str(validation_frame.iloc[i,0]))\n        for j in range(1,7): \n            print(\"predicition = \" +  str(validation_frame.iloc[i,j]) )\n        for k in range(7,13): \n            print(\"true predicition = \" +  str(validation_frame.iloc[i,k]))\n# activation map","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.689185Z","iopub.execute_input":"2021-07-22T14:53:20.689486Z","iopub.status.idle":"2021-07-22T14:53:20.708654Z","shell.execute_reply.started":"2021-07-22T14:53:20.689451Z","shell.execute_reply":"2021-07-22T14:53:20.707878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"any_preds = validation_frame['any']\nmax_index = any_preds.idxmax()\nmax_index","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.709891Z","iopub.execute_input":"2021-07-22T14:53:20.710221Z","iopub.status.idle":"2021-07-22T14:53:20.718792Z","shell.execute_reply.started":"2021-07-22T14:53:20.710187Z","shell.execute_reply":"2021-07-22T14:53:20.717532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_to_heatmap(): \n    highest_predicted_img = validation_frame.loc[max_index,'filename']\n    if validation_frame.loc[max_index, 'true_any'] == 1:\n        return highest_predicted_img","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.720393Z","iopub.execute_input":"2021-07-22T14:53:20.720827Z","iopub.status.idle":"2021-07-22T14:53:20.725865Z","shell.execute_reply.started":"2021-07-22T14:53:20.720793Z","shell.execute_reply":"2021-07-22T14:53:20.724456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"highest_predicted_img =  img_to_heatmap()\nhighest_predicted_img","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:53:20.72725Z","iopub.execute_input":"2021-07-22T14:53:20.727632Z","iopub.status.idle":"2021-07-22T14:53:20.735625Z","shell.execute_reply.started":"2021-07-22T14:53:20.727588Z","shell.execute_reply":"2021-07-22T14:53:20.734707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:53:42.859255Z","iopub.execute_input":"2021-07-23T17:53:42.85961Z","iopub.status.idle":"2021-07-23T17:53:42.876515Z","shell.execute_reply.started":"2021-07-23T17:53:42.859576Z","shell.execute_reply":"2021-07-23T17:53:42.875437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2021-07-23T17:54:04.56828Z","iopub.execute_input":"2021-07-23T17:54:04.568655Z","iopub.status.idle":"2021-07-23T17:54:04.589501Z","shell.execute_reply.started":"2021-07-23T17:54:04.568622Z","shell.execute_reply":"2021-07-23T17:54:04.588314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_list_test = []\nfor pred in test_preds: \n    predictions_list_test.append(pred)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.178116Z","iopub.execute_input":"2021-07-22T14:54:49.178527Z","iopub.status.idle":"2021-07-22T14:54:49.197552Z","shell.execute_reply.started":"2021-07-22T14:54:49.178483Z","shell.execute_reply":"2021-07-22T14:54:49.196808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_frame =  test_sample_df.drop(['filename'], axis=1)\ntest_frame","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.198676Z","iopub.execute_input":"2021-07-22T14:54:49.199028Z","iopub.status.idle":"2021-07-22T14:54:49.219813Z","shell.execute_reply.started":"2021-07-22T14:54:49.198994Z","shell.execute_reply":"2021-07-22T14:54:49.218847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_frame.iloc[:,:] = predictions_list_test\ntest_frame","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.223113Z","iopub.execute_input":"2021-07-22T14:54:49.223421Z","iopub.status.idle":"2021-07-22T14:54:49.278912Z","shell.execute_reply.started":"2021-07-22T14:54:49.223394Z","shell.execute_reply":"2021-07-22T14:54:49.277706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sample_df = test_sample_df.stack().reset_index()\ntest_sample_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T15:00:54.75971Z","iopub.execute_input":"2021-07-22T15:00:54.760092Z","iopub.status.idle":"2021-07-22T15:00:54.840998Z","shell.execute_reply.started":"2021-07-22T15:00:54.76005Z","shell.execute_reply":"2021-07-22T15:00:54.839899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_frame.insert(0, \"filename\", test_df.filename)\ntest_frame","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.280522Z","iopub.execute_input":"2021-07-22T14:54:49.280982Z","iopub.status.idle":"2021-07-22T14:54:49.313497Z","shell.execute_reply.started":"2021-07-22T14:54:49.280937Z","shell.execute_reply":"2021-07-22T14:54:49.312605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nfor i in range(20): \n  \n    for j in range(1,7): \n        if test_frame.iloc[i,j] > 0.8: \n            path = '/kaggle/tmp/' + str(test_frame.iloc[i,0])\n            img = Image.open(path)\n            plt.imshow(img)\n            print(str(test_frame.iloc[i,0]) + \" has a probability: \"  + str(test_frame.iloc[i,j]) + \" for a '\" + str(test_frame.columns[j]) + \"' type of hemorrhage\")\n            plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.314762Z","iopub.execute_input":"2021-07-22T14:54:49.315101Z","iopub.status.idle":"2021-07-22T14:54:49.737723Z","shell.execute_reply.started":"2021-07-22T14:54:49.315057Z","shell.execute_reply":"2021-07-22T14:54:49.736847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#heatmap \n#The code used to show the heatmake was taken from: https://keras.io/examples/vision/grad_cam/\n#Only slightly modified to fit this workflow and return the image with the highest predicition from the validtion set \nfrom IPython.display import Image, display\n\npreprocess_input = keras.applications.xception.preprocess_input\ndecode_predictions = keras.applications.xception.decode_predictions\n\nlast_conv_layer_name = \"block14_sepconv2_act\"\n\nimg_path = '/kaggle/tmp/' + str(highest_predicted_img)\ndisplay(Image(img_path))","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.743293Z","iopub.execute_input":"2021-07-22T14:54:49.743536Z","iopub.status.idle":"2021-07-22T14:54:49.753134Z","shell.execute_reply.started":"2021-07-22T14:54:49.743512Z","shell.execute_reply":"2021-07-22T14:54:49.752298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img_array(img_path, size):\n    # `img` is a PIL image of size 299x299\n    img = keras.preprocessing.image.load_img(img_path, target_size=size)\n    # `array` is a float32 Numpy array of shape (299, 299, 3)\n    array = keras.preprocessing.image.img_to_array(img)\n    # We add a dimension to transform our array into a \"batch\"\n    # of size (1, 299, 299, 3)\n    array = np.expand_dims(array, axis=0)\n    return array\n\ndef make_gradcam_heatmap(img_array, model, last_conv_layer_name, pred_index=None):\n    # First, we create a model that maps the input image to the activations\n    # of the last conv layer as well as the output predictions\n    grad_model = tf.keras.models.Model(\n        [model.inputs], [model.get_layer(last_conv_layer_name).output, model.output]\n    )\n\n    # Then, we compute the gradient of the top predicted class for our input image\n    # with respect to the activations of the last conv layer\n    with tf.GradientTape() as tape:\n        last_conv_layer_output, preds = grad_model(img_array)\n        if pred_index is None:\n            pred_index = tf.argmax(preds[0])\n        class_channel = preds[:, pred_index]\n\n    # This is the gradient of the output neuron (top predicted or chosen)\n    # with regard to the output feature map of the last conv layer\n    grads = tape.gradient(class_channel, last_conv_layer_output)\n\n    # This is a vector where each entry is the mean intensity of the gradient\n    # over a specific feature map channel\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n\n    # We multiply each channel in the feature map array\n    # by \"how important this channel is\" with regard to the top predicted class\n    # then sum all the channels to obtain the heatmap class activation\n    last_conv_layer_output = last_conv_layer_output[0]\n    heatmap = last_conv_layer_output @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n\n    # For visualization purpose, we will also normalize the heatmap between 0 & 1\n    heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap)\n    return heatmap.numpy()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.75449Z","iopub.execute_input":"2021-07-22T14:54:49.754844Z","iopub.status.idle":"2021-07-22T14:54:49.76427Z","shell.execute_reply.started":"2021-07-22T14:54:49.754811Z","shell.execute_reply":"2021-07-22T14:54:49.763402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size = (299, 299)\n\nimg_array = preprocess_input(get_img_array(img_path, size=img_size))\n\nmodel.layers[-1].activation = None\n\nheatmap = make_gradcam_heatmap(img_array, model, last_conv_layer_name)\n\n# Display heatmap\nplt.matshow(heatmap)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:49.76547Z","iopub.execute_input":"2021-07-22T14:54:49.765987Z","iopub.status.idle":"2021-07-22T14:54:50.179679Z","shell.execute_reply.started":"2021-07-22T14:54:49.765952Z","shell.execute_reply":"2021-07-22T14:54:50.178716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.cm as cm\ndef save_and_display_gradcam(img_path, heatmap, cam_path=\"cam.jpg\", alpha=0.4):\n    # Load the original image\n    img = keras.preprocessing.image.load_img(img_path)\n    img = keras.preprocessing.image.img_to_array(img)\n\n    # Rescale heatmap to a range 0-255\n    heatmap = np.uint8(255 * heatmap)\n\n    # Use jet colormap to colorize heatmap\n    jet = cm.get_cmap(\"jet\")\n\n    # Use RGB values of the colormap\n    jet_colors = jet(np.arange(256))[:, :3]\n    jet_heatmap = jet_colors[heatmap]\n\n    # Create an image with RGB colorized heatmap\n    jet_heatmap = keras.preprocessing.image.array_to_img(jet_heatmap)\n    jet_heatmap = jet_heatmap.resize((img.shape[1], img.shape[0]))\n    jet_heatmap = keras.preprocessing.image.img_to_array(jet_heatmap)\n\n    # Superimpose the heatmap on original image\n    superimposed_img = jet_heatmap * alpha + img\n    superimposed_img = keras.preprocessing.image.array_to_img(superimposed_img)\n\n    # Save the superimposed image\n    superimposed_img.save(cam_path)\n\n    # Display Grad CAM\n    display(Image(cam_path))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:50.180927Z","iopub.execute_input":"2021-07-22T14:54:50.181476Z","iopub.status.idle":"2021-07-22T14:54:50.195122Z","shell.execute_reply.started":"2021-07-22T14:54:50.181436Z","shell.execute_reply":"2021-07-22T14:54:50.194229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_and_display_gradcam(img_path, heatmap)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T14:54:50.196484Z","iopub.execute_input":"2021-07-22T14:54:50.19701Z","iopub.status.idle":"2021-07-22T14:54:50.225431Z","shell.execute_reply.started":"2021-07-22T14:54:50.196973Z","shell.execute_reply":"2021-07-22T14:54:50.22457Z"},"trusted":true},"execution_count":null,"outputs":[]}]}