{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- Table of contents:\n     1. Read Data\n     2. Preprocessing\n     3. CNN Modeling\n     4. Predicting and Submitting\n-----------------\n# 1. Read Data","metadata":{}},{"cell_type":"code","source":"# Importing neccessary packages\nimport os\nimport time\nimport gc\nimport math\nimport tifffile as tifi\nimport cv2\nimport PIL\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom PIL import Image, ImageOps\nfrom openslide import OpenSlide\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Conv2D, MaxPooling2D, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras.utils import image_dataset_from_directory\nfrom sklearn.model_selection import train_test_split\n\n# Ignore the warning\nimport warnings\nwarnings.filterwarnings('ignore')\n#------------------------------------------\nnotebook_start_time = time.time()","metadata":{"execution":{"iopub.status.busy":"2022-09-24T23:31:19.188724Z","iopub.execute_input":"2022-09-24T23:31:19.189171Z","iopub.status.idle":"2022-09-24T23:31:25.48671Z","shell.execute_reply.started":"2022-09-24T23:31:19.189091Z","shell.execute_reply":"2022-09-24T23:31:25.485743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read CSV files\ntrain_df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest_df = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ntest_df[\"file_path\"]  = test_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/test/\" + x + \".tif\")\ntrain_df[\"file_path\"]  = train_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\ntrain_df['target'] = np.where(train_df['label'] =='CE', int(1), int(0))\ntrain_df['size_gb'] = train_df['file_path'].apply(lambda x: np.round(os.path.getsize(x)/1000000000,2))\ntest_df['size_gb'] = test_df['file_path'].apply(lambda x: np.round(os.path.getsize(x)/1000000000,2))\ntrain_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-09-24T23:31:25.488837Z","iopub.execute_input":"2022-09-24T23:31:25.489493Z","iopub.status.idle":"2022-09-24T23:31:26.709595Z","shell.execute_reply.started":"2022-09-24T23:31:25.489453Z","shell.execute_reply":"2022-09-24T23:31:26.708739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Skip this part if you don't want a headache lol.\n# But generally, this code is to read tiff files,\n# resize them and appending them to a list.\ndef resize_with_padding(img, expected_size):\n    img.thumbnail((expected_size[0], expected_size[1]))\n    delta_width = expected_size[0] - img.size[0]\n    delta_height = expected_size[1] - img.size[1]\n    pad_width = delta_width // 2\n    pad_height = delta_height // 2\n    padding = (pad_width, pad_height, delta_width - pad_width, delta_height - pad_height)\n    return ImageOps.expand(img, padding,fill=0)\n###-----------------------------\n\ntrain_img_list = []\ny_list = []\ndef train_tif_to_array(df):\n    time_list = []\n    for i in range(len(df)):\n        size = np.array(df.loc[:,'size_gb'])[i]\n        if size >0.75:\n            start_time = time.time()\n            if i !=0:\n                if i % 5==0: \n                    print('processed {} of large .tif'.format(i))\n                    print('Time per image:',(sum(time_list)/len(time_list)), 'seconds')\n            y_list.append(int(np.array(train_df.loc[:,'target'])[i]))\n            slide = OpenSlide(np.array(df.loc[:,'file_path'])[i])\n            #sf = 64\n            img_size = (27000,27000)\n            #level = slide.get_best_level_for_downsample(sf)\n            image = np.array(slide.read_region((250,250), 0, img_size,).resize((265, 265)).convert('RGB'))\n            th = 10 \n            black_pixels = np.where(\n            (image[:, :, 0] < th) & \n            (image[:, :, 1] < th) & \n            (image[:, :, 2] < th))\n            image[black_pixels] = [255, 255, 255]\n            train_img_list.append(np.array(image))\n            del image\n            del black_pixels\n            time_list.append(np.round((time.time() - start_time), 2))\n            for i in range(2):\n                gc.collect()\n            \n        else:\n            start_time = time.time()\n            if i !=0:\n                if i % 5==0: \n                    print('processed {} of mid .tif'.format(i))\n                    print('Time per image:',(sum(time_list)/len(time_list)), 'seconds')\n            LOAD_TRUNCATED_IMAGES = True\n            PIL.Image.MAX_IMAGE_PIXELS =None\n            y_list.append(int(np.array(train_df.loc[:,'target'])[i]))\n            image = ImageOps.contain(PIL.Image.open(np.array(df.loc[:,'file_path'])[i]), (265,265,3))\n            image = image.rotate(90, PIL.Image.NEAREST, expand = 1)\n            image = resize_with_padding(image, (265, 265))\n\n            rr, gg, bb = image.split()\n            rr = rr.point(lambda p: 255 if p==0 else p)\n            gg = gg.point(lambda p: 255 if p==0 else p)\n            bb = bb.point(lambda p: 255 if p==0 else p)\n            image = Image.merge(\"RGB\", (rr, gg, bb))\n            train_img_list.append(np.array(image))\n            del image\n            del rr\n            del gg\n            del bb\n            time_list.append(np.round((time.time() - start_time), 2))\n            for i in range(2):\n                gc.collect()\n    del time_list\n####---------------------------\n# I couldn't make them in one function, due to the labels,\n# which the test files lack.\ntest_img_list = []\ndef test_tif_to_array(df):\n    for i in range(len(df)):\n        size = np.array(df.loc[:,'size_gb'])[i]\n        if size >0.8:\n            if i % 1==0: \n                print('processed {} of large .tif'.format(i)) \n            slide = OpenSlide(np.array(df.loc[:,'file_path'])[i])\n            #sf = 64\n            img_size = (27000,27000)\n            #level = slide.get_best_level_for_downsample(sf)\n            image = np.array(slide.read_region((50,50), 0, img_size,).resize((265, 265)).convert('RGB'))\n            th = 10 \n            black_pixels = np.where(\n            (image[:, :, 0] < th) & \n            (image[:, :, 1] < th) & \n            (image[:, :, 2] < th))\n            image[black_pixels] = [255, 255, 255]\n            test_img_list.append(np.array(image))\n            del image\n            del black_pixels\n            for i in range(2):\n                gc.collect()\n\n        else:\n            if i % 1==0: \n                print('processed {} of mid .tif'.format(i)) \n            LOAD_TRUNCATED_IMAGES = True\n            PIL.Image.MAX_IMAGE_PIXELS =None\n            image = ImageOps.contain(PIL.Image.open(np.array(df.loc[:,'file_path'])[i]), (265,265,3))\n            image = image.rotate(90, PIL.Image.NEAREST, expand = 1)\n            image = resize_with_padding(image, (265, 265))\n\n            rr, gg, bb = image.split()\n            rr = rr.point(lambda p: 255 if p==0 else p)\n            gg = gg.point(lambda p: 255 if p==0 else p)\n            bb = bb.point(lambda p: 255 if p==0 else p)\n            image = Image.merge(\"RGB\", (rr, gg, bb))\n            test_img_list.append(np.array(image))\n            del image\n            del rr\n            del gg\n            del bb\n            for i in range(2):\n                gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-24T23:31:26.711047Z","iopub.execute_input":"2022-09-24T23:31:26.711622Z","iopub.status.idle":"2022-09-24T23:31:26.738337Z","shell.execute_reply.started":"2022-09-24T23:31:26.711585Z","shell.execute_reply":"2022-09-24T23:31:26.73682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Preprocessing\n- Preprocessing is estimated to take about 7:(+-10mins) hours, let's hope the estimation is wrong because the notebook will terminate automatically after 9 hours.","metadata":{}},{"cell_type":"code","source":"test_tif_to_array(test_df)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-24T23:31:46.639006Z","iopub.execute_input":"2022-09-24T23:31:46.639382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tif_to_array(train_df) ","metadata":{"execution":{"iopub.status.busy":"2022-09-18T10:25:00.667039Z","iopub.execute_input":"2022-09-18T10:25:00.667438Z","iopub.status.idle":"2022-09-18T10:28:41.828467Z","shell.execute_reply.started":"2022-09-18T10:25:00.667405Z","shell.execute_reply":"2022-09-18T10:28:41.827001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the files into validation and train, so we can monitor \n# the model's ability to generalize. Test size is about 1 picture per 100 train,\n# meaning we have about 7-8, since we have about 754 train data files.\ny_train = np.array(y_list)\ntrain_img_list = np.array(train_img_list)\ntest_img_list = np.array(test_img_list)\n(X_train, X_val, y_train, y_val) = train_test_split(train_img_list, y_train, test_size=0.01, random_state=42)\ndel train_img_list\ndel y_list\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Please note that all the files that will be in test are unique, meaning they're not included in the train files; to make sure the model performs well on unseen data.","metadata":{}},{"cell_type":"markdown","source":"------------------------------------\n# 3. CNN Modeling","metadata":{}},{"cell_type":"code","source":"# Cleaning up extra memory.\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using data augmentation; due to the small size of our train files.\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rotation_range=25,\n    brightness_range=[0.9,1.1],\n    shear_range=0.05,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    vertical_flip=True,\n     preprocessing_function=tf.keras.applications.nasnet.preprocess_input, \n)\n\nval_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.nasnet.preprocess_input\n)\n#----------------------------------------\n\ntrain_generator = train_datagen.flow(X_train, y_train, batch_size=32)\nval_generator = val_datagen.flow(X_val, y_val, batch_size=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# .fit Callbacks\nearly_stopping_monitor = EarlyStopping(monitor='val_loss', patience=70, restore_best_weights=True)\n\nbest_model = ModelCheckpoint('bestmodel.hdf5', monitor='val_loss', save_best_only=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CNN Modeling..\nmodel = Sequential()\ninput_shape = (265, 265, 3)\n\n\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), padding = 'same', activation = 'relu', input_shape = input_shape))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(BatchNormalization()),\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(BatchNormalization()),\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(filters=256, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(BatchNormalization()),\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(filters=128, kernel_size = (3,3), padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters=256, kernel_size = (3,3), padding = 'same', activation = 'relu',kernel_regularizer=tf.keras.regularizers.l2(0.0001)))\nmodel.add(BatchNormalization()),\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(filters=512, kernel_size = (3,3), activation = 'relu'))\n#model.add(BatchNormalization()), # may//maynot\n\nmodel.add(Flatten())\nmodel.add(Dense(1024, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l1(0.000000015)))\nmodel.add(Dropout(0.22))\nmodel.add(Dense(512, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l1_l2(0.0000000151)))\nmodel.add(Dense(256, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l1(0.0000000153)))\nmodel.add(Dropout(0.22))   \nmodel.add(Dense(128, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l1(0.0000000152)))\nmodel.add(Dropout(0.22))\n#model.add(BatchNormalization()),  # Was used before\nmodel.add(Dense(64, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l1(0.0000000153)))\nmodel.add(Dense(32, activation = 'relu',kernel_regularizer=tf.keras.regularizers.l1(0.0000000153)))\n\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(\n    loss = tf.keras.losses.MeanSquaredError(),    \n    metrics=['accuracy'],\n    optimizer = tf.keras.optimizers.RMSprop(0.00004, momentum=0.6))\nmodel.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning up extra memory.\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model..\nstart_time = time.time()\nmodel.fit_generator(train_generator,\n         validation_data=val_generator,\n         epochs=210, workers=8, max_queue_size=10,\n         callbacks=[best_model, early_stopping_monitor])\nprint('\\n------------\\nTime took:', np.round((time.time() - start_time), 2))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accuracy of our model..\nplt.figure(figsize=(12, 6))\nplt.plot(model.history.history[\"accuracy\"])\nplt.plot(model.history.history[\"val_accuracy\"])\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.title(\"ACCURACY OF MODEL\")\nplt.legend(['training_accuracy', 'validation_accuracy'])\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loss of our model..\nplt.figure(figsize=(12, 6))\nplt.plot(model.history.history[\"loss\"])\nplt.plot(model.history.history[\"val_loss\"])\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.title(\"LOSS OF MODEL\")\nplt.legend(['training_loss', 'validation_loss', 'past_loss'])\nplt.show()\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Due to the small size of our validation files, our val_loss and our val_accuracy function shouldn't be taken so seriously. \n---------------------------------\n# 4. Predicting and Submitting","metadata":{}},{"cell_type":"code","source":"# Loading our model's best weights, generally on val_loss.\nmodel.load_weights('bestmodel.hdf5')\nmodel.evaluate(val_generator);","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Functions to predict the test images.\n# -- Note that these functions automate predictions,\n# -- so that predictions are correctly classified to each image.\ntest_img_list = np.array(test_img_list)\ntest_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.nasnet.preprocess_input\n)\ndef predict(i):\n    test_generator = test_datagen.flow(x=test_img_list[i:i+1], batch_size=1)\n    return model.predict(test_generator)\ndef make_predictions(df):\n    for i in range(len(df[0:])):\n        if i ==0:\n            print('Predicting test images..\\n-----------\\nPrediction results:')\n            cnn_pred = predict(i)\n        else:\n            cnn_pred = np.vstack((cnn_pred,predict(i)))\n    return cnn_pred","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting...\ncnn_pred = make_predictions(test_img_list)\nprint(cnn_pred)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Appending predictions to the csv...\nsubmission = pd.DataFrame(test_df[\"patient_id\"].copy())\nsubmission[\"CE\"] = cnn_pred\nsubmission[\"LAA\"] = 1- submission[\"CE\"]\n\nsubmission = submission.groupby(\"patient_id\").mean()\nsubmission = submission[[\"CE\", \"LAA\"]].round(6).reset_index()\nsubmission","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submitting...\nsubmission.to_csv(\"submission.csv\", index = False)\ntime.sleep(5)\n!head submission.csv","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Notebook took about:\\n', np.round((time.time() - notebook_start_time)/3600, 2), 'hours')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----------------------\n- To be honest, that took me more than a week of consistent tough hardwork and headaches too :) , because the model was almost always stuck at a local minima, plus I had to understand how tiff files work so I can preprocess it in the fastest way possible without using much ram since the ram is less when using an accelerator like a GPU for instance and fast because the notebook's termination is set to 9 hours by default, add to that that I had to understand why the model wasn't predicting correctly, turns out giving the model each image alone was the best solution.\n    \n    Thankfully, after a lot of experimenting, a good model was found, a good tiff file converter was coded, and a good function to automate predictions was also coded. \n    \n- Please note that no additional data was used, nor was internet used, it was all keras, tensorflow and various other packages to preprocess the data, anyway, I hope my hardwork contributes to a good cause afterall. Thanks for following me till the end, you're clearly an awesome one!","metadata":{}}]}