{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install mlflow","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-05T06:27:40.375506Z","iopub.execute_input":"2021-09-05T06:27:40.375787Z","iopub.status.idle":"2021-09-05T06:27:56.529863Z","shell.execute_reply.started":"2021-09-05T06:27:40.37572Z","shell.execute_reply":"2021-09-05T06:27:56.528855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %env SM_FRAMEWORK=tf.keras","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:27:56.532924Z","iopub.execute_input":"2021-09-05T06:27:56.533326Z","iopub.status.idle":"2021-09-05T06:27:56.539859Z","shell.execute_reply.started":"2021-09-05T06:27:56.533284Z","shell.execute_reply":"2021-09-05T06:27:56.539087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlflow.tensorflow\nmlflow.tensorflow.autolog()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:27:56.542222Z","iopub.execute_input":"2021-09-05T06:27:56.542699Z","iopub.status.idle":"2021-09-05T06:28:01.340392Z","shell.execute_reply.started":"2021-09-05T06:27:56.542663Z","shell.execute_reply":"2021-09-05T06:28:01.339555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport os\nimport collections\nimport sys\nimport glob\nimport random\nimport cv2\nimport tensorflow as tf\nimport multiprocessing\n\nfrom math import ceil, floor\nfrom copy import deepcopy\nfrom tqdm import tqdm_notebook as tqdm\nfrom imgaug import augmenters as iaa\n\nimport tensorflow.keras\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.callbacks import Callback, ModelCheckpoint\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.losses import binary_crossentropy\nfrom tensorflow.keras.optimizers import Adam\n\ndef calculating_class_weights(y_true):\n    from sklearn.utils.class_weight import compute_class_weight\n    number_dim = np.shape(y_true)[1]\n    weights = np.empty([number_dim, 2])\n    for i in range(number_dim):\n        weights[i] = compute_class_weight('balanced', [0.,1.], y_true[:, i])\n    return weights","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:01.342057Z","iopub.execute_input":"2021-09-05T06:28:01.342401Z","iopub.status.idle":"2021-09-05T06:28:03.070901Z","shell.execute_reply.started":"2021-09-05T06:28:01.342366Z","shell.execute_reply":"2021-09-05T06:28:03.069803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet\n!pip install iterative-stratification","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:03.074159Z","iopub.execute_input":"2021-09-05T06:28:03.074434Z","iopub.status.idle":"2021-09-05T06:28:17.522776Z","shell.execute_reply.started":"2021-09-05T06:28:03.074408Z","shell.execute_reply":"2021-09-05T06:28:17.52182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Custom Modules\nimport efficientnet.tfkeras as efn \nfrom iterstrat.ml_stratifiers import MultilabelStratifiedShuffleSplit","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:17.52445Z","iopub.execute_input":"2021-09-05T06:28:17.524809Z","iopub.status.idle":"2021-09-05T06:28:17.677516Z","shell.execute_reply.started":"2021-09-05T06:28:17.524769Z","shell.execute_reply":"2021-09-05T06:28:17.676656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed\nSEED = 12345\nnp.random.seed(SEED)\n# tf.set_random_seed(SEED)\n\n# Constants\nTEST_SIZE = 0.1\nHEIGHT = 256\nWIDTH = 256\nCHANNELS = 3\nTRAIN_BATCH_SIZE = 32\nVALID_BATCH_SIZE = 64\nSHAPE = (HEIGHT, WIDTH, CHANNELS)\n\n# Folders\nDATA_DIR = '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/'\nTEST_IMAGES_DIR = DATA_DIR + 'stage_2_test/'\nTRAIN_IMAGES_DIR = DATA_DIR + 'stage_2_train/'","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:17.678745Z","iopub.execute_input":"2021-09-05T06:28:17.67912Z","iopub.status.idle":"2021-09-05T06:28:17.686339Z","shell.execute_reply.started":"2021-09-05T06:28:17.67905Z","shell.execute_reply":"2021-09-05T06:28:17.684371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correct_dcm(dcm):\n    x = dcm.pixel_array + 1000\n    px_mode = 4096\n    x[x>=px_mode] = x[x>=px_mode] - px_mode\n    dcm.PixelData = x.tobytes()\n    dcm.RescaleIntercept = -1000\n\ndef window_image(dcm, window_center, window_width):    \n    if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n        correct_dcm(dcm)\n    img = dcm.pixel_array * dcm.RescaleSlope + dcm.RescaleIntercept\n    \n    # Resize\n    img = cv2.resize(img, SHAPE[:2], interpolation = cv2.INTER_LINEAR)\n   \n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    img = np.clip(img, img_min, img_max)\n    return img\n\ndef bsb_window(dcm0,dcm1,dcm2):\n    brain_img = window_image(dcm0, 40, 380)\n    subdural_img = window_image(dcm1, 40, 380)\n    soft_img = window_image(dcm2, 40, 380)\n    \n    brain_img = (brain_img - (-150)) / 380\n    subdural_img = (subdural_img - (-150)) / 380\n    soft_img = (soft_img - (-150)) / 380\n    bsb_img = np.array([brain_img, subdural_img, soft_img]).transpose(1,2,0)\n    return bsb_img\n\ndef _read(path0,path1,path2, SHAPE):\n    dcm0 = pydicom.dcmread(path0)\n    dcm1= pydicom.dcmread(path1)\n    dcm2 = pydicom.dcmread(path2)\n    \n    \n    try:\n        img = bsb_window(dcm0,dcm1,dcm2)\n    except:\n#         img = bsb_window(dcm)\n        img = np.zeros(SHAPE)\n    return img","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:17.689857Z","iopub.execute_input":"2021-09-05T06:28:17.690282Z","iopub.status.idle":"2021-09-05T06:28:17.70383Z","shell.execute_reply.started":"2021-09-05T06:28:17.690247Z","shell.execute_reply":"2021-09-05T06:28:17.702441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image Augmentation\nsometimes = lambda aug: iaa.Sometimes(0.25, aug)\naugmentation = iaa.Sequential([ iaa.Fliplr(0.25),\n                                iaa.Flipud(0.10),\n                                sometimes(iaa.Crop(px=(0, 25), keep_size = True, sample_independently = False))   \n                            ], random_order = True)       \n        \n# Generators\nclass TrainDataGenerator(tensorflow.keras.utils.Sequence):\n    def __init__(self, dataset, labels, batch_size = 16, img_size = SHAPE, img_dir = TRAIN_IMAGES_DIR, augment = False, *args, **kwargs):\n        self.dataset = dataset\n        self.ids = dataset.index\n#         self.len_ids = len(self.ids)\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_dir = img_dir\n        self.augment = augment\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(ceil(len(self.ids) / self.batch_size))\n\n    def __getitem__(self, index):\n        indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        X, Y = self.__data_generation(indices)\n        return X, Y\n\n    def augmentor(self, image):\n        augment_img = augmentation        \n        image_aug = augment_img.augment_image(image)\n        return image_aug\n\n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.ids))\n#         np.random.shuffle(self.indices)\n\n    def __data_generation(self, indices):\n        X = np.empty((self.batch_size, *self.img_size))\n        Y = np.empty((self.batch_size, 6), dtype=np.float32)\n        \n        for i, index in enumerate(indices):\n            ID0 = self.ids[index-1]\n            ID1 = self.ids[index]\n            try:\n                ID2 = self.ids[index+1]\n            except:\n                ID2 = self.ids[index]\n            \n            \n            image = _read(self.img_dir+'ID_'+ID0+\".dcm\",self.img_dir+'ID_'+ID1+\".dcm\",self.img_dir+'ID_'+ID2+\".dcm\", self.img_size)\n            \n            if self.augment:\n                X[i,] = self.augmentor(image)\n            else:\n                X[i,] = image\n            Y[i,] = self.labels.iloc[index].values        \n        return X, Y\n\nclass ValDataGenerator(tensorflow.keras.utils.Sequence):\n    def __init__(self, dataset, labels, batch_size = 16, img_size = SHAPE, img_dir = TRAIN_IMAGES_DIR, augment = False, *args, **kwargs):\n        self.dataset = dataset\n        self.ids = dataset.index\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_dir = img_dir\n        self.augment = augment\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(ceil(len(self.ids) / self.batch_size))\n\n    def __getitem__(self, index):\n        indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        X, Y = self.__data_generation(indices)\n        return X, Y\n\n    def augmentor(self, image):\n        augment_img = augmentation        \n        image_aug = augment_img.augment_image(image)\n        return image_aug\n\n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.ids))\n\n    def __data_generation(self, indices):\n        X = np.empty((self.batch_size, *self.img_size))\n        Y = np.empty((self.batch_size, 6), dtype=np.float32)\n        \n        for i, index in enumerate(indices):\n            ID = self.ids[index]\n            image = _read(self.img_dir+ID+\".dcm\", self.img_size)\n            if self.augment:\n                X[i,] = self.augmentor(image)\n            else:\n                X[i,] = image\n            Y[i,] = self.labels.iloc[index].values        \n        return X, Y\n    \nclass TestDataGenerator(tensorflow.keras.utils.Sequence):\n    def __init__(self, dataset, labels, batch_size = 16, img_size = SHAPE, img_dir = TEST_IMAGES_DIR, *args, **kwargs):\n        self.dataset = dataset\n        self.ids = dataset.index\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_dir = img_dir\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(ceil(len(self.ids) / self.batch_size))\n\n    def __getitem__(self, index):\n        indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        X = self.__data_generation(indices)\n        return X\n\n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.ids))\n    \n    def __data_generation(self, indices):\n        X = np.empty((self.batch_size, *self.img_size))\n        \n        for i, index in enumerate(indices):\n            ID = self.ids[index]\n            image = _read(self.img_dir+ID+\".dcm\", self.img_size)\n            X[i,] = image              \n        return X","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:17.707413Z","iopub.execute_input":"2021-09-05T06:28:17.707692Z","iopub.status.idle":"2021-09-05T06:28:17.736922Z","shell.execute_reply.started":"2021-09-05T06:28:17.707668Z","shell.execute_reply":"2021-09-05T06:28:17.736017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_testset(filename = DATA_DIR + \"stage_2_sample_submission.csv\"):\n    df = pd.read_csv(filename)\n    df[\"Image\"] = df[\"ID\"].str.slice(stop=12)\n    df[\"Diagnosis\"] = df[\"ID\"].str.slice(start=13)\n    df = df.loc[:, [\"Label\", \"Diagnosis\", \"Image\"]]\n    df = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n    return df\n\ndef read_trainset(filename = DATA_DIR + \"stage_2_train.csv\"):\n    df = pd.read_csv(filename)\n    df[\"Image\"] = df[\"ID\"].str.slice(stop=12)\n    df[\"Diagnosis\"] = df[\"ID\"].str.slice(start=13)\n    duplicates_to_remove = [56346, 56347, 56348, 56349,\n                            56350, 56351, 1171830, 1171831,\n                            1171832, 1171833, 1171834, 1171835,\n                            3705312, 3705313, 3705314, 3705315,\n                            3705316, 3705317, 3842478, 3842479,\n                            3842480, 3842481, 3842482, 3842483 ]\n    df = df.drop(index = duplicates_to_remove)\n    df = df.reset_index(drop = True)    \n    df = df.loc[:, [\"Label\", \"Diagnosis\", \"Image\"]]\n    df = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n    return df\n\n# Read Train and Test Datasets\ntest_df = read_testset()\ntrain_df = read_trainset()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:17.738484Z","iopub.execute_input":"2021-09-05T06:28:17.739171Z","iopub.status.idle":"2021-09-05T06:28:34.580872Z","shell.execute_reply.started":"2021-09-05T06:28:17.739126Z","shell.execute_reply":"2021-09-05T06:28:34.580001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.iloc[:]\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:34.58211Z","iopub.execute_input":"2021-09-05T06:28:34.58247Z","iopub.status.idle":"2021-09-05T06:28:34.735861Z","shell.execute_reply.started":"2021-09-05T06:28:34.582434Z","shell.execute_reply":"2021-09-05T06:28:34.734884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Oversampling\nepidural_df = train_df[train_df.Label['epidural'] == 1]\ntrain_oversample_df = pd.concat([train_df, epidural_df])\ntrain_df = train_oversample_df\n\n# Summary\nprint('Train Shape: {}'.format(train_df.shape))\nprint('Test Shape: {}'.format(test_df.shape))","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:34.737346Z","iopub.execute_input":"2021-09-05T06:28:34.737793Z","iopub.status.idle":"2021-09-05T06:28:34.854548Z","shell.execute_reply.started":"2021-09-05T06:28:34.737755Z","shell.execute_reply":"2021-09-05T06:28:34.853564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_f = pd.read_csv('../input/intracranial-hemorrhage-seresnext50-v2/meta_patient_homorrhage.csv')\nnp.sum(data_f.values[:,[6,1,2,3,4,5]],axis=0)/len(data_f)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:34.855815Z","iopub.execute_input":"2021-09-05T06:28:34.856361Z","iopub.status.idle":"2021-09-05T06:28:37.996014Z","shell.execute_reply.started":"2021-09-05T06:28:34.856322Z","shell.execute_reply":"2021-09-05T06:28:37.995161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_f.index = data_f['id']\ndata_f","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:37.997399Z","iopub.execute_input":"2021-09-05T06:28:37.997758Z","iopub.status.idle":"2021-09-05T06:28:38.052553Z","shell.execute_reply.started":"2021-09-05T06:28:37.99772Z","shell.execute_reply":"2021-09-05T06:28:38.051511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_f.iloc[:75999]","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:38.053926Z","iopub.execute_input":"2021-09-05T06:28:38.054411Z","iopub.status.idle":"2021-09-05T06:28:38.094083Z","shell.execute_reply.started":"2021-09-05T06:28:38.054355Z","shell.execute_reply":"2021-09-05T06:28:38.09314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train = data_f.iloc[75999:,[6,1,2,3,4,5]].copy()\nnew_val   = data_f.iloc[:75999,[6,1,2,3,4,5]].copy()\n\nnew_train","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:38.095532Z","iopub.execute_input":"2021-09-05T06:28:38.095918Z","iopub.status.idle":"2021-09-05T06:28:38.144168Z","shell.execute_reply.started":"2021-09-05T06:28:38.095878Z","shell.execute_reply":"2021-09-05T06:28:38.143227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_val","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:38.145636Z","iopub.execute_input":"2021-09-05T06:28:38.146011Z","iopub.status.idle":"2021-09-05T06:28:38.166168Z","shell.execute_reply.started":"2021-09-05T06:28:38.145973Z","shell.execute_reply":"2021-09-05T06:28:38.165067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = calculating_class_weights(data_f.values[:,[6,1,2,3,4,5]])\nweights","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:38.167804Z","iopub.execute_input":"2021-09-05T06:28:38.168244Z","iopub.status.idle":"2021-09-05T06:28:41.027447Z","shell.execute_reply.started":"2021-09-05T06:28:38.168204Z","shell.execute_reply":"2021-09-05T06:28:41.026405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predictions(test_df, model):    \n    test_preds = model.predict_generator(TestDataGenerator(test_df, None, 8, SHAPE, TEST_IMAGES_DIR), verbose = 1)\n    return test_preds[:test_df.iloc[range(test_df.shape[0])].shape[0]]\n\ndef ModelCheckpointFull(model_name):\n    return ModelCheckpoint(model_name, \n                            monitor = 'val_AUC_full', \n                            verbose = 1, \n                            save_best_only = True, \n                            save_weights_only = True, \n                            mode = 'max', \n                            period = 1)\n\n# Create Model\ndef create_model():\n    K.clear_session()\n    \n    base_model =  efn.EfficientNetB2(weights = 'imagenet', include_top = False, pooling = 'avg', input_shape = SHAPE)\n    x = base_model.output\n    x = Dropout(0.15)(x)\n    y_pred = Dense(6, activation = 'sigmoid')(x)\n\n    return Model(inputs = base_model.input, outputs = y_pred)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.028956Z","iopub.execute_input":"2021-09-05T06:28:41.029352Z","iopub.status.idle":"2021-09-05T06:28:41.036968Z","shell.execute_reply.started":"2021-09-05T06:28:41.029315Z","shell.execute_reply":"2021-09-05T06:28:41.035914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Submission Placeholder\n# submission_predictions = []\n\n# # Multi Label Stratified Split stuff...\n# msss = MultilabelStratifiedShuffleSplit(n_splits = 10, test_size = TEST_SIZE, random_state = SEED)\n# X = train_df.index\n# Y = train_df.Label.values","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.038385Z","iopub.execute_input":"2021-09-05T06:28:41.038717Z","iopub.status.idle":"2021-09-05T06:28:41.052703Z","shell.execute_reply.started":"2021-09-05T06:28:41.038683Z","shell.execute_reply":"2021-09-05T06:28:41.051901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Get train and test index\n# msss_splits = next(msss.split(X, Y))\n# train_idx = msss_splits[0]\n# valid_idx = msss_splits[1]","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.053997Z","iopub.execute_input":"2021-09-05T06:28:41.054433Z","iopub.status.idle":"2021-09-05T06:28:41.062382Z","shell.execute_reply.started":"2021-09-05T06:28:41.054397Z","shell.execute_reply":"2021-09-05T06:28:41.061297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(train_idx),len(valid_idx)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.065423Z","iopub.execute_input":"2021-09-05T06:28:41.06573Z","iopub.status.idle":"2021-09-05T06:28:41.070912Z","shell.execute_reply.started":"2021-09-05T06:28:41.065682Z","shell.execute_reply":"2021-09-05T06:28:41.070014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.074941Z","iopub.execute_input":"2021-09-05T06:28:41.075253Z","iopub.status.idle":"2021-09-05T06:28:41.093682Z","shell.execute_reply.started":"2021-09-05T06:28:41.075218Z","shell.execute_reply":"2021-09-05T06:28:41.092933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.random.shuffle(train_idx)\n# print(train_idx[:5])    \n# print(valid_idx[:5])\n\n# data_generator_train = TrainDataGenerator(train_df.iloc[train_idx], \n#                                             train_df.iloc[train_idx], \n#                                             TRAIN_BATCH_SIZE, \n#                                             SHAPE,\n#                                             augment = True)\n","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.095366Z","iopub.execute_input":"2021-09-05T06:28:41.095886Z","iopub.status.idle":"2021-09-05T06:28:41.10034Z","shell.execute_reply.started":"2021-09-05T06:28:41.095849Z","shell.execute_reply":"2021-09-05T06:28:41.099378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.101746Z","iopub.execute_input":"2021-09-05T06:28:41.102098Z","iopub.status.idle":"2021-09-05T06:28:41.129961Z","shell.execute_reply.started":"2021-09-05T06:28:41.102063Z","shell.execute_reply":"2021-09-05T06:28:41.129238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator_train = TrainDataGenerator(new_train.iloc[:], \n                                           new_train.iloc[:], \n                                            TRAIN_BATCH_SIZE, \n                                            SHAPE,\n                                            augment = True)\n\ndata_generator_val = TrainDataGenerator(new_val.iloc[:], \n                                           new_val.iloc[:], \n                                            VALID_BATCH_SIZE, \n                                            SHAPE,\n                                            augment = False)\n\nTRAIN_STEPS = int(len(data_generator_train) / 10)\nLR = 0.0001\n\n\n\nfor i, j in data_generator_val:\n    print(i.shape)\n    break","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:41.154213Z","iopub.execute_input":"2021-09-05T06:28:41.154564Z","iopub.status.idle":"2021-09-05T06:28:43.213659Z","shell.execute_reply.started":"2021-09-05T06:28:41.154531Z","shell.execute_reply":"2021-09-05T06:28:43.212676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_STEPS","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:43.215056Z","iopub.execute_input":"2021-09-05T06:28:43.215455Z","iopub.status.idle":"2021-09-05T06:28:43.221782Z","shell.execute_reply.started":"2021-09-05T06:28:43.215401Z","shell.execute_reply":"2021-09-05T06:28:43.220469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.imshow(i[5,:,:,0])\nplt.colorbar()\nplt.show()\nplt.imshow(i[5,:,:,1])\nplt.show()\nplt.imshow(i[5,:,:,2])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:43.223483Z","iopub.execute_input":"2021-09-05T06:28:43.224099Z","iopub.status.idle":"2021-09-05T06:28:43.707983Z","shell.execute_reply.started":"2021-09-05T06:28:43.224061Z","shell.execute_reply":"2021-09-05T06:28:43.707173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# (train_df.iloc[valid_idx].values[:,5]==1).sum()/10","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:43.709403Z","iopub.execute_input":"2021-09-05T06:28:43.7099Z","iopub.status.idle":"2021-09-05T06:28:43.713869Z","shell.execute_reply.started":"2021-09-05T06:28:43.709863Z","shell.execute_reply":"2021-09-05T06:28:43.713043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUC = tf.keras.metrics.AUC\nRECALL = tf.keras.metrics.Recall\nPRECISION = tf.keras.metrics.Precision\n# PRECISION = tf.keras.metrics.RecallAtPrecision","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:43.715332Z","iopub.execute_input":"2021-09-05T06:28:43.715831Z","iopub.status.idle":"2021-09-05T06:28:43.724783Z","shell.execute_reply.started":"2021-09-05T06:28:43.715783Z","shell.execute_reply":"2021-09-05T06:28:43.7239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Model\nMetrics = [AUC(name = 'AUC_full', multi_label=True),\n           AUC(name = 'AUC_0', multi_label=True, label_weights=[1,0,0,0,0,0]),\n           AUC(name = 'AUC_1', multi_label=True, label_weights=[0,1,0,0,0,0]),\n           AUC(name = 'AUC_2', multi_label=True, label_weights=[0,0,1,0,0,0]),\n           AUC(name = 'AUC_3', multi_label=True, label_weights=[0,0,0,1,0,0]),\n           AUC(name = 'AUC_4', multi_label=True, label_weights=[0,0,0,0,1,0]),\n           AUC(name = 'AUC_5', multi_label=True, label_weights=[0,0,0,0,0,1]),\n           \n           RECALL(thresholds=0.7,name='REC_full'),\n           RECALL(thresholds=0.7,class_id=0, name='REC_0'),\n           RECALL(thresholds=0.7,class_id=1, name='REC_1'),\n           RECALL(thresholds=0.7,class_id=2, name='REC_2'),\n           RECALL(thresholds=0.7,class_id=3, name='REC_3'),\n           RECALL(thresholds=0.7,class_id=4, name='REC_4'),\n           RECALL(thresholds=0.7,class_id=5, name='REC_5')]\n          \n#            PRECISION(thresholds=0.7, name='PRE_full'),\n#            PRECISION(thresholds=0.7, class_id=0, name='PRE_0'),\n#            PRECISION(thresholds=0.7, class_id=1, name='PRE_1'),\n#            PRECISION(thresholds=0.7, class_id=2, name='PRE_2'),\n#            PRECISION(thresholds=0.7, class_id=3, name='PRE_3'),\n#            PRECISION(thresholds=0.7, class_id=4, name='PRE_4'),\n#            PRECISION(thresholds=0.7, class_id=5, name='PRE_5')]\n\ndef get_weighted_loss(weights):\n    def weighted_loss(y_true, y_pred):\n        return K.mean((weights[:,0]**(1-y_true))*(weights[:,1]**(y_true))*K.binary_crossentropy(y_true, y_pred), axis=-1)\n    return weighted_loss\n\nmodel = create_model()   \nmodel.compile(optimizer = Adam(learning_rate = LR), \n                  loss = get_weighted_loss(weights),\n                  metrics = Metrics)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:43.72595Z","iopub.execute_input":"2021-09-05T06:28:43.726404Z","iopub.status.idle":"2021-09-05T06:28:49.148533Z","shell.execute_reply.started":"2021-09-05T06:28:43.726317Z","shell.execute_reply":"2021-09-05T06:28:49.147714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.load_weights('../input/hemorrhageefficientnetb2v2best-weights/model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:49:45.506394Z","iopub.execute_input":"2021-09-04T23:49:45.506721Z","iopub.status.idle":"2021-09-04T23:49:45.511651Z","shell.execute_reply.started":"2021-09-04T23:49:45.506687Z","shell.execute_reply":"2021-09-04T23:49:45.510718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install tensorflow-gpu==2.5.0","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:49:45.513066Z","iopub.execute_input":"2021-09-04T23:49:45.513623Z","iopub.status.idle":"2021-09-04T23:49:45.525496Z","shell.execute_reply.started":"2021-09-04T23:49:45.513585Z","shell.execute_reply":"2021-09-04T23:49:45.524581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def main():\nwith mlflow.start_run():\n    model.fit_generator(generator = data_generator_train,\n                            validation_data = data_generator_val,\n                            steps_per_epoch = TRAIN_STEPS,\n                            epochs = 10,\n                            callbacks = [ModelCheckpointFull('model.h5')],\n                            verbose = 1,workers=4)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:49.1498Z","iopub.execute_input":"2021-09-05T06:28:49.150125Z","iopub.status.idle":"2021-09-05T06:29:41.969991Z","shell.execute_reply.started":"2021-09-05T06:28:49.15009Z","shell.execute_reply":"2021-09-05T06:29:41.961301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# res = model.evaluate_generator(data_generator_val,\n#                             verbose = 1,workers=4)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:52:11.271021Z","iopub.status.idle":"2021-09-04T23:52:11.273094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_generator_val = ValDataGenerator(train_df.iloc[valid_idx], \n#                                         train_df.iloc[valid_idx], \n#                                         VALID_BATCH_SIZE, \n#                                         SHAPE,\n#                                         augment = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:25:16.284566Z","iopub.status.idle":"2021-09-04T23:25:16.285103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result = model.predict_generator(data_generator_val,\n#                             verbose = 1,workers=4)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:25:16.286199Z","iopub.status.idle":"2021-09-04T23:25:16.286962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # y_true = []\n# k = 0\n# for i,j in data_generator_val:\n#     print(k)\n#     if k==0:\n#         y_true = np.array(j)\n#     else:       \n#         y_true = np.concatenate((y_true,j),axis=0)\n#     k+=1","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:25:16.288185Z","iopub.status.idle":"2021-09-04T23:25:16.288815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result.shape,y_true.shape[0]/(1181+1)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:25:16.289969Z","iopub.status.idle":"2021-09-04T23:25:16.290594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# roc_auc_score(np.array(np.concatenate((y_true[:64,0],y_true[128:,0]),axis=0),int), result[:,i])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T23:20:16.472873Z","iopub.status.idle":"2021-09-04T23:20:16.473655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# from sklearn.metrics import precision_recall_curve,roc_auc_score\n# print(y_true.shape)\n# i = 0\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n# precision, recall, thresholds","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:46:48.862Z","iopub.execute_input":"2021-08-04T12:46:48.862349Z","iopub.status.idle":"2021-08-04T12:46:48.886097Z","shell.execute_reply.started":"2021-08-04T12:46:48.862316Z","shell.execute_reply":"2021-08-04T12:46:48.885182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i=0\n# roc_auc_score(np.array(y_true[:,i],int), result[:,i])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# keys = train_df['Label'].columns\n# keys","metadata":{"execution":{"iopub.status.busy":"2021-09-04T22:04:31.372274Z","iopub.execute_input":"2021-09-04T22:04:31.372615Z","iopub.status.idle":"2021-09-04T22:04:31.379753Z","shell.execute_reply.started":"2021-09-04T22:04:31.372582Z","shell.execute_reply":"2021-09-04T22:04:31.378801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# plt.figure(figsize=(8,8))\n# plt.style.use('seaborn')\n# i = 0\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n# # plt.subplot(2,3,i+1)\n# plt.xlabel('Recall')\n# plt.ylabel('Precision')\n\n# plt.plot(recall,precision,label=f'{keys[i]}')\n\n# i = 1\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n# # plt.subplot(2,3,i+1)\n# plt.xlabel('Recall')\n# plt.ylabel('Precision')\n\n# plt.plot(recall,precision,label=f'{keys[i]}')\n\n# i = 2\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n# # plt.subplot(2,3,i+1)\n# plt.xlabel('Recall')\n# plt.ylabel('Precision')\n\n# plt.plot(recall,precision,label=f'{keys[i]}')\n\n# i = 3\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n# # plt.subplot(2,3,i+1)\n# plt.xlabel('Recall')\n# plt.ylabel('Precision')\n\n# plt.plot(recall,precision,label=f'{keys[i]}')\n\n# i = 4\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n# # plt.subplot(2,3,i+1)\n# plt.xlabel('Recall')\n# plt.ylabel('Precision')\n\n# plt.plot(recall,precision,label=f'{keys[i]}')\n\n# i = 5\n# precision, recall, thresholds = precision_recall_curve(np.array(y_true[:,i],int), result[:,i])\n\n# plt.xlabel('Recall')\n# plt.ylabel('Precision')\n\n# plt.plot(recall,precision,label=f'{keys[i]}')\n# plt.legend()\n\n# plt.savefig('pr_re_curve.jpg',dpi=250,bbox_inches='tight')","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:23:04.629508Z","iopub.execute_input":"2021-08-04T13:23:04.629827Z","iopub.status.idle":"2021-08-04T13:23:06.045532Z","shell.execute_reply.started":"2021-08-04T13:23:04.629797Z","shell.execute_reply":"2021-08-04T13:23:06.044697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def main():\n# with mlflow.start_run():\n#     model.fit_generator(generator = data_generator_train,\n#                             validation_data = data_generator_val,\n#                             steps_per_epoch = TRAIN_STEPS,\n#                             epochs = 17,\n#                             callbacks = [ModelCheckpointFull('model.h5')],\n#                             verbose = 1,workers=4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.mean([0.978,0.989,0.985,0.992,0.968,0.967]),0.98","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:35:48.762305Z","iopub.execute_input":"2021-08-04T13:35:48.762629Z","iopub.status.idle":"2021-08-04T13:35:48.76827Z","shell.execute_reply.started":"2021-08-04T13:35:48.7626Z","shell.execute_reply":"2021-08-04T13:35:48.767471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}