{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4696088,"sourceType":"datasetVersion","datasetId":2687741}],"dockerImageVersionId":30356,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import needed modules","metadata":{"id":"CKeVGxZ5GG6o"}},{"cell_type":"code","source":"# import system libs\nimport os\nimport time\nimport shutil\nimport pathlib\nimport itertools\nfrom pathlib import Path\nimport multiprocessing as mp\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\n\n# import data handling tools\nimport cv2\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport matplotlib.pyplot as plt\n\n# import Deep learning Libraries\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Activation, Dropout, BatchNormalization\nfrom tensorflow.keras.models import Model, load_model, Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam, Adamax\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras.metrics import categorical_crossentropy\n\n# Ignore Warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nprint ('modules loaded')","metadata":{"id":"CeMcAy_5GG6s","execution":{"iopub.status.busy":"2024-03-02T15:26:53.344093Z","iopub.execute_input":"2024-03-02T15:26:53.344463Z","iopub.status.idle":"2024-03-02T15:27:00.125515Z","shell.execute_reply.started":"2024-03-02T15:26:53.344384Z","shell.execute_reply":"2024-03-02T15:27:00.124478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create needed functions","metadata":{"id":"SA_gwvwnGG6v"}},{"cell_type":"markdown","source":"## Function to create dataframe\nWe will use the _create_df()_ function to create train and validation dataframe depending on _define_trpaths()_ and _define_trdf()_ functions which are responsible for getting file paths.","metadata":{"id":"JQdhl_CRGG6v"}},{"cell_type":"code","source":"def define_trpaths(train_data, train_csv):\n\n    filepaths = []\n    labels = []\n    df = pd.read_csv(train_csv)\n    files = os.listdir(train_data)\n\n    for file, i in zip(sorted(files), df['cancer']):\n        foldpath = os.path.join(train_data, file)\n        files = os.listdir(foldpath)\n\n        for f in files:\n            fpath = os.path.join(foldpath, f)\n            filepaths.append(fpath)\n            if i == 0:\n                labels.append('No Cancer')\n            elif i == 1:\n                labels.append('Cancer')\n                \n    return filepaths, labels\n\ndef define_trdf(files, classes):\n    Fseries = pd.Series(files, name= 'filepaths')\n    Lseries = pd.Series(classes, name='labels')\n    return pd.concat([Fseries, Lseries], axis= 1)\n\ndef create_df(train_data, train_csv):\n    # train dataframe\n    files, classes = define_trpaths(train_data, train_csv)\n    df = define_trdf(files, classes)\n    strat = df['labels']\n    train_df, valid_df = train_test_split(df, train_size= 0.7, shuffle= True, random_state= 123, stratify= strat)\n    \n    return train_df, valid_df","metadata":{"id":"La4bEbHlGG6w","execution":{"iopub.status.busy":"2024-03-02T15:27:14.079929Z","iopub.execute_input":"2024-03-02T15:27:14.080582Z","iopub.status.idle":"2024-03-02T15:27:14.091968Z","shell.execute_reply.started":"2024-03-02T15:27:14.080545Z","shell.execute_reply":"2024-03-02T15:27:14.090857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function to trim data samples\nwe will use a function *trim()* that takes in a dataframe df, and integer max_size and a string column and returns a dataframe where the number of samples for any class specified by column is limited to max samples.","metadata":{}},{"cell_type":"code","source":"def trim (df, max_size, min_size, column):\n    df = df.copy()\n    original_class_count = len(list(df[column].unique()))\n    print ('Original Number of classes in dataframe: ', original_class_count)\n    \n    sample_list = [] \n    groups = df.groupby(column)\n    \n    for label in df[column].unique():        \n        group = groups.get_group(label)\n        sample_count = len(group)     \n        if sample_count > max_size :\n            strat = group[column]\n            samples, _ = train_test_split(group, train_size= max_size, shuffle= True, \n                                          random_state= 123, stratify= strat)            \n            sample_list.append(samples)\n      \n        elif sample_count >= min_size:\n            sample_list.append(group)\n    \n    df = pd.concat(sample_list, axis= 0).reset_index(drop= True)\n    final_class_count = len(list(df[column].unique())) \n    if final_class_count != original_class_count:\n        print ('*** WARNING***  dataframe has a reduced number of classes' )\n    balance = list(df[column].value_counts())\n    print (balance)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:27:19.632704Z","iopub.execute_input":"2024-03-02T15:27:19.633131Z","iopub.status.idle":"2024-03-02T15:27:19.643806Z","shell.execute_reply.started":"2024-03-02T15:27:19.633098Z","shell.execute_reply":"2024-03-02T15:27:19.64268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function to generate images from dataframe\n_create_gens()_ function is responsible for generating batches of tensor image data with real-time data augmentation.","metadata":{"id":"JZaHdeFxGG6x"}},{"cell_type":"code","source":"def create_gens(train_df, valid_df, train_dir):\n    img_size = (224, 224)\n    channels = 3\n    batch_size = 40\n    img_shape = (img_size[0], img_size[1], channels)\n    \n    def scalar(img):\n        return img\n    tr_gen = ImageDataGenerator(preprocessing_function= scalar, horizontal_flip= True)\n    val_gen = ImageDataGenerator(preprocessing_function= scalar)\n    train_gen = tr_gen.flow_from_dataframe( train_df, x_col= 'filepaths', y_col= 'labels', directory = train_dir, target_size= img_size, class_mode= 'categorical',\n                                        color_mode= 'rgb', shuffle= True, batch_size= batch_size)\n    valid_gen = val_gen.flow_from_dataframe( valid_df, x_col= 'filepaths', y_col= 'labels', directory = train_dir, target_size= img_size, class_mode= 'categorical',\n                                        color_mode= 'rgb', shuffle= True, batch_size= batch_size)\n\n    return train_gen, valid_gen","metadata":{"id":"iLL8hHQcGG6x","execution":{"iopub.status.busy":"2024-03-02T15:27:20.632423Z","iopub.execute_input":"2024-03-02T15:27:20.633177Z","iopub.status.idle":"2024-03-02T15:27:20.644096Z","shell.execute_reply.started":"2024-03-02T15:27:20.633131Z","shell.execute_reply":"2024-03-02T15:27:20.643008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function to show images\n*show_images()* function is responsible for showing images sample from a specific directory after image generator.","metadata":{"id":"8ifXox4SGG6y"}},{"cell_type":"code","source":"def show_images(gen):\n    g_dict = gen.class_indices\n    classes = list(g_dict.keys())\n    images, labels = next(gen)\n    plt.figure(figsize= (20, 20))\n    length = len(labels)\n    sample = min(length, 25)\n    for i in range(sample):\n        plt.subplot(5, 5, i + 1)\n        image = images[i] / 255\n        plt.imshow(image)\n        index = np.argmax(labels[i])\n        class_name = classes[index]\n        plt.title(class_name, color= 'blue', fontsize= 12)\n        plt.axis('off')\n    plt.show()","metadata":{"id":"IAGbj3ZyGG6y","execution":{"iopub.status.busy":"2024-03-02T15:27:21.573957Z","iopub.execute_input":"2024-03-02T15:27:21.574586Z","iopub.status.idle":"2024-03-02T15:27:21.584717Z","shell.execute_reply.started":"2024-03-02T15:27:21.574539Z","shell.execute_reply":"2024-03-02T15:27:21.583718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Callback Class\n* We will use a custom callback class MyCallback which is responsible for modifying hyperparameters in run-time.\n\n* It will inherit its parameters and hyperparameters from keras.callbacks.Callback.","metadata":{"id":"_K-ryg0DGG6z"}},{"cell_type":"code","source":"### Define a class for custom callback\nclass MyCallback(keras.callbacks.Callback):\n    def __init__(self, model, base_model, patience, stop_patience, threshold, factor, batches, initial_epoch, epochs, ask_epoch):\n        super(MyCallback, self).__init__()\n        self.model = model\n        self.base_model = base_model\n        self.patience = patience # specifies how many epochs without improvement before learning rate is adjusted\n        self.stop_patience = stop_patience # specifies how many times to adjust lr without improvement to stop training\n        self.threshold = threshold # specifies training accuracy threshold when lr will be adjusted based on validation loss\n        self.factor = factor # factor by which to reduce the learning rate\n        self.batches = batches # number of training batch to runn per epoch\n        self.initial_epoch = initial_epoch\n        self.epochs = epochs\n        self.ask_epoch = ask_epoch\n        self.ask_epoch_initial = ask_epoch # save this value to restore if restarting training\n        # callback variables\n        self.count = 0 # how many times lr has been reduced without improvement\n        self.stop_count = 0\n        self.best_epoch = 1   # epoch with the lowest loss\n        self.initial_lr = float(tf.keras.backend.get_value(model.optimizer.lr)) # get the initial learning rate and save it\n        self.highest_tracc = 0.0 # set highest training accuracy to 0 initially\n        self.lowest_vloss = np.inf # set lowest validation loss to infinity initially\n        self.best_weights = self.model.get_weights() # set best weights to model's initial weights\n        self.initial_weights = self.model.get_weights()   # save initial weights if they have to get restored\n\n    # Define a function that will run when train begins\n    def on_train_begin(self, logs= None):\n        msg = '{0:^8s}{1:^10s}{2:^9s}{3:^9s}{4:^9s}{5:^9s}{6:^9s}{7:^10s}{8:10s}{9:^8s}'.format('Epoch', 'Loss', 'Accuracy', 'V_loss', 'V_acc', 'LR', 'Next LR', 'Monitor','% Improv', 'Duration')\n        print(msg)\n        self.start_time = time.time()\n\n    def on_train_end(self, logs= None):\n        stop_time = time.time()\n        tr_duration = stop_time - self.start_time\n        hours = tr_duration // 3600\n        minutes = (tr_duration - (hours * 3600)) // 60\n        seconds = tr_duration - ((hours * 3600) + (minutes * 60))\n        msg = f'training elapsed time was {str(hours)} hours, {minutes:4.1f} minutes, {seconds:4.2f} seconds)'\n        print(msg)\n        self.model.set_weights(self.best_weights) # set the weights of the model to the best weights\n\n    def on_train_batch_end(self, batch, logs= None):\n        acc = logs.get('accuracy') * 100 # get batch accuracy\n        loss = logs.get('loss')\n        msg = '{0:20s}processing batch {1:} of {2:5s}-   accuracy=  {3:5.3f}   -   loss: {4:8.5f}'.format(' ', str(batch), str(self.batches), acc, loss)\n        print(msg, '\\r', end= '') # prints over on the same line to show running batch count\n\n    def on_epoch_begin(self, epoch, logs= None):\n        self.ep_start = time.time()\n\n    # Define method runs on the end of each epoch\n    def on_epoch_end(self, epoch, logs= None):\n        ep_end = time.time()\n        duration = ep_end - self.ep_start\n\n        lr = float(tf.keras.backend.get_value(self.model.optimizer.lr)) # get the current learning rate\n        current_lr = lr\n        acc = logs.get('accuracy')  # get training accuracy\n        v_acc = logs.get('val_accuracy')  # get validation accuracy\n        loss = logs.get('loss')  # get training loss for this epoch\n        v_loss = logs.get('val_loss')  # get the validation loss for this epoch\n\n        if acc < self.threshold: # if training accuracy is below threshold adjust lr based on training accuracy\n            monitor = 'accuracy'\n            if epoch == 0:\n                pimprov = 0.0\n            else:\n                pimprov = (acc - self.highest_tracc ) * 100 / self.highest_tracc # define improvement of model progres\n\n            if acc > self.highest_tracc: # training accuracy improved in the epoch\n                self.highest_tracc = acc # set new highest training accuracy\n                self.best_weights = self.model.get_weights() # training accuracy improved so save the weights\n                self.count = 0 # set count to 0 since training accuracy improved\n                self.stop_count = 0 # set stop counter to 0\n                if v_loss < self.lowest_vloss:\n                    self.lowest_vloss = v_loss\n                self.best_epoch = epoch + 1  # set the value of best epoch for this epoch\n\n            else:\n                # training accuracy did not improve check if this has happened for patience number of epochs\n                # if so adjust learning rate\n                if self.count >= self.patience - 1: # lr should be adjusted\n                    lr = lr * self.factor # adjust the learning by factor\n                    tf.keras.backend.set_value(self.model.optimizer.lr, lr) # set the learning rate in the optimizer\n                    self.count = 0 # reset the count to 0\n                    self.stop_count = self.stop_count + 1 # count the number of consecutive lr adjustments\n                    self.count = 0 # reset counter\n                    if v_loss < self.lowest_vloss:\n                        self.lowest_vloss = v_loss\n                else:\n                    self.count = self.count + 1 # increment patience counter\n\n        else: # training accuracy is above threshold so adjust learning rate based on validation loss\n            monitor = 'val_loss'\n            if epoch == 0:\n                pimprov = 0.0\n            else:\n                pimprov = (self.lowest_vloss - v_loss ) * 100 / self.lowest_vloss\n            if v_loss < self.lowest_vloss: # check if the validation loss improved\n                self.lowest_vloss = v_loss # replace lowest validation loss with new validation loss\n                self.best_weights = self.model.get_weights() # validation loss improved so save the weights\n                self.count = 0 # reset count since validation loss improved\n                self.stop_count = 0\n                self.best_epoch = epoch + 1 # set the value of the best epoch to this epoch\n            else: # validation loss did not improve\n                if self.count >= self.patience - 1: # need to adjust lr\n                    lr = lr * self.factor # adjust the learning rate\n                    self.stop_count = self.stop_count + 1 # increment stop counter because lr was adjusted\n                    self.count = 0 # reset counter\n                    tf.keras.backend.set_value(self.model.optimizer.lr, lr) # set the learning rate in the optimizer\n                else:\n                    self.count = self.count + 1 # increment the patience counter\n                if acc > self.highest_tracc:\n                    self.highest_tracc = acc\n\n        msg = f'{str(epoch + 1):^3s}/{str(self.epochs):4s} {loss:^9.3f}{acc * 100:^9.3f}{v_loss:^9.5f}{v_acc * 100:^9.3f}{current_lr:^9.5f}{lr:^9.5f}{monitor:^11s}{pimprov:^10.2f}{duration:^8.2f}'\n        print(msg)\n\n        if self.stop_count > self.stop_patience - 1: # check if learning rate has been adjusted stop_count times with no improvement\n            msg = f' training has been halted at epoch {epoch + 1} after {self.stop_patience} adjustments of learning rate with no improvement'\n            print(msg)\n            self.model.stop_training = True # stop training","metadata":{"id":"d5HiN8XDGG60","execution":{"iopub.status.busy":"2024-03-02T15:27:24.158234Z","iopub.execute_input":"2024-03-02T15:27:24.158655Z","iopub.status.idle":"2024-03-02T15:27:24.19143Z","shell.execute_reply.started":"2024-03-02T15:27:24.158624Z","shell.execute_reply":"2024-03-02T15:27:24.190215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function to plot history of training\nWe will use plot_training() function for plotting trainning history items [accuracy and loss] in train data and validation data.","metadata":{"id":"2zwhoj3zGG61"}},{"cell_type":"code","source":"def plot_training(hist):\n    tr_acc = hist.history['accuracy']\n    tr_loss = hist.history['loss']\n    val_acc = hist.history['val_accuracy']\n    val_loss = hist.history['val_loss']\n    index_loss = np.argmin(val_loss)\n    val_lowest = val_loss[index_loss]\n    index_acc = np.argmax(val_acc)\n    acc_highest = val_acc[index_acc]\n\n    plt.figure(figsize= (20, 8))\n    plt.style.use('fivethirtyeight')\n    Epochs = [i+1 for i in range(len(tr_acc))]\n    loss_label = f'best epoch= {str(index_loss + 1)}'\n    acc_label = f'best epoch= {str(index_acc + 1)}'\n    plt.subplot(1, 2, 1)\n    plt.plot(Epochs, tr_loss, 'r', label= 'Training loss')\n    plt.plot(Epochs, val_loss, 'g', label= 'Validation loss')\n    plt.scatter(index_loss + 1, val_lowest, s= 150, c= 'blue', label= loss_label)\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.subplot(1, 2, 2)\n    plt.plot(Epochs, tr_acc, 'r', label= 'Training Accuracy')\n    plt.plot(Epochs, val_acc, 'g', label= 'Validation Accuracy')\n    plt.scatter(index_acc + 1 , acc_highest, s= 150, c= 'blue', label= acc_label)\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    plt.tight_layout\n    plt.show()\n","metadata":{"id":"pU3eAW5jGG62","execution":{"iopub.status.busy":"2024-03-02T15:27:25.530469Z","iopub.execute_input":"2024-03-02T15:27:25.530859Z","iopub.status.idle":"2024-03-02T15:27:25.54323Z","shell.execute_reply.started":"2024-03-02T15:27:25.530829Z","shell.execute_reply":"2024-03-02T15:27:25.541965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Structure","metadata":{"id":"57eDFl3oGG65"}},{"cell_type":"markdown","source":"### Show images sample","metadata":{"id":"2GHNMVrhGG65"}},{"cell_type":"code","source":"# Get Dataframes\ntrain_data = '/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_dicomsdl_512/train_images_processed_cv2_dicomsdl_512'\ntrain_csv = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\n\ntrain_df, valid_df = create_df(train_data, train_csv)\ntrain_df = trim(train_df, max_size = 10000, min_size = 0, column='labels')\nvalid_df = trim(valid_df, max_size = 5000, min_size = 0, column='labels')\n# Get Generators\ntrain_gen, valid_gen = create_gens(train_df, valid_df, train_data)\n\nshow_images(train_gen)","metadata":{"id":"1xfcIPMeGG65","execution":{"iopub.status.busy":"2024-03-02T15:27:28.197925Z","iopub.execute_input":"2024-03-02T15:27:28.198657Z","iopub.status.idle":"2024-03-02T15:30:09.990719Z","shell.execute_reply.started":"2024-03-02T15:27:28.198624Z","shell.execute_reply":"2024-03-02T15:30:09.989608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create Pre-trained model\nWe will use _EfficientNetB5","metadata":{"id":"3wvOKjeRGG65"}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adamax\nfrom tensorflow.keras import regularizers\n\n# Define image size and number of channels\nimg_size = (224, 224)\nchannels = 3\nimg_shape = (img_size[0], img_size[1], channels)\n\n# Define the number of classes\nclass_count = len(train_gen.class_indices)\n\n# Load the pre-trained EfficientNetB3 model\nbase_model = tf.keras.applications.EfficientNetB3(\n    include_top=False, weights='imagenet', input_shape=img_shape, pooling='max')\n\n# Create the model architecture\nmodel = Sequential([\n    base_model,\n    BatchNormalization(axis=-1, momentum=0.99, epsilon=0.001),\n    Dense(256, kernel_regularizer=regularizers.l2(l=0.016), \n          activity_regularizer=regularizers.l1(0.006),\n          bias_regularizer=regularizers.l1(0.006), activation='relu'),\n    Dropout(rate=0.45, seed=123),\n    Dense(class_count, activation='softmax')\n])\n\n# Compile the model\nmodel.compile(Adamax(learning_rate=0.001), \n              loss='categorical_crossentropy', \n              metrics=['accuracy'])\n\n# Print the model summary\nmodel.summary()\n","metadata":{"id":"e0JI_Zd_GG66","execution":{"iopub.status.busy":"2024-03-02T15:39:29.527838Z","iopub.execute_input":"2024-03-02T15:39:29.528633Z","iopub.status.idle":"2024-03-02T15:39:51.631515Z","shell.execute_reply.started":"2024-03-02T15:39:29.528597Z","shell.execute_reply":"2024-03-02T15:39:51.63007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get custom callbacks parameters","metadata":{"id":"TciwhdM1GG66"}},{"cell_type":"code","source":"batch_size = 40\nepochs = 20\npatience = 1 \t\t# number of epochs to wait to adjust lr if monitored value does not improve\nstop_patience = 3 \t# number of epochs to wait before stopping training if monitored value does not improve\nthreshold = 0.9 \t# if train accuracy is < threshhold adjust monitor accuracy, else monitor validation loss\nfactor = 0.5 \t\t# factor to reduce lr by\nfreeze = False \t\t# if true free weights of  the base model\nask_epoch = 5\t\t# number of epochs to run before asking if you want to halt training\nbatches = int(np.ceil(len(train_gen.labels) / batch_size))\n\ncallbacks = [MyCallback(model= model, base_model= base_model, patience= patience,\n            stop_patience= stop_patience, threshold= threshold, factor= factor,\n            batches= batches, initial_epoch= 0, epochs= epochs, ask_epoch= ask_epoch )]","metadata":{"id":"7abvdv7mGG66","execution":{"iopub.status.busy":"2023-01-18T20:26:06.21977Z","iopub.execute_input":"2023-01-18T20:26:06.22175Z","iopub.status.idle":"2023-01-18T20:26:06.517035Z","shell.execute_reply.started":"2023-01-18T20:26:06.22172Z","shell.execute_reply":"2023-01-18T20:26:06.516008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train model","metadata":{"id":"ap89fjdxGG67"}},{"cell_type":"code","source":"history = model.fit(x= train_gen, epochs= epochs, verbose= 0, callbacks= callbacks,\n                    validation_data= valid_gen, validation_steps= None, shuffle= False,\n                    initial_epoch= 0)","metadata":{"id":"0Uk3BTERGG67","execution":{"iopub.status.busy":"2023-01-18T20:26:06.518434Z","iopub.execute_input":"2023-01-18T20:26:06.518856Z","iopub.status.idle":"2023-01-18T21:04:00.602191Z","shell.execute_reply.started":"2023-01-18T20:26:06.518818Z","shell.execute_reply":"2023-01-18T21:04:00.601194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot training history","metadata":{"id":"dNKq6ebOGG67"}},{"cell_type":"code","source":"plot_training(history)","metadata":{"id":"L0Bj0Sp_GG68","execution":{"iopub.status.busy":"2023-01-18T21:04:00.603685Z","iopub.execute_input":"2023-01-18T21:04:00.604039Z","iopub.status.idle":"2023-01-18T21:04:01.098933Z","shell.execute_reply.started":"2023-01-18T21:04:00.604003Z","shell.execute_reply":"2023-01-18T21:04:01.097916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluate model","metadata":{"id":"MySXhfAJGG68"}},{"cell_type":"code","source":"train_score = model.evaluate(train_gen, verbose= 1)\nvalid_score = model.evaluate(valid_gen, verbose= 1)\n\nprint(\"Train Loss: \", train_score[0])\nprint(\"Train Accuracy: \", train_score[1])\nprint('-' * 20)\nprint(\"Validation Loss: \", valid_score[0])\nprint(\"Validation Accuracy: \", valid_score[1])","metadata":{"id":"wSKDkyXXGG68","execution":{"iopub.status.busy":"2023-01-18T21:04:01.100536Z","iopub.execute_input":"2023-01-18T21:04:01.100908Z","iopub.status.idle":"2023-01-18T21:05:08.887715Z","shell.execute_reply.started":"2023-01-18T21:04:01.100872Z","shell.execute_reply":"2023-01-18T21:05:08.886554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convert Dicoms to PNG","metadata":{}},{"cell_type":"code","source":"def convert_images(filename, outdir):\n    ds = pydicom.read_file(str(filename))\n    img = ds.pixel_array\n    img = cv2.resize(img, (128, 128))\n    cv2.imwrite(outdir + filename.split('/')[-1][:-4] + '.png', img)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:05:08.950587Z","iopub.execute_input":"2023-01-18T21:05:08.950929Z","iopub.status.idle":"2023-01-18T21:05:08.9569Z","shell.execute_reply.started":"2023-01-18T21:05:08.950896Z","shell.execute_reply":"2023-01-18T21:05:08.955568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '/kaggle/input/rsna-breast-cancer-detection/test_images'\ntest_out_path = '/kaggle/working/test_png/'","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:05:08.958557Z","iopub.execute_input":"2023-01-18T21:05:08.959008Z","iopub.status.idle":"2023-01-18T21:05:08.968268Z","shell.execute_reply.started":"2023-01-18T21:05:08.958973Z","shell.execute_reply":"2023-01-18T21:05:08.967177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(test_out_path):\n    os.makedirs(test_out_path)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:05:08.969609Z","iopub.execute_input":"2023-01-18T21:05:08.970415Z","iopub.status.idle":"2023-01-18T21:05:08.980329Z","shell.execute_reply.started":"2023-01-18T21:05:08.970381Z","shell.execute_reply":"2023-01-18T21:05:08.979409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\ntest_dcm_list = glob.glob(os.path.join(test_path, '**/*.dcm'))","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:05:08.981413Z","iopub.execute_input":"2023-01-18T21:05:08.981786Z","iopub.status.idle":"2023-01-18T21:05:08.997397Z","shell.execute_reply.started":"2023-01-18T21:05:08.981749Z","shell.execute_reply":"2023-01-18T21:05:08.996478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res2 = Parallel(n_jobs=8, backend='threading')(delayed(\n    convert_images)(i, test_out_path) for i in tqdm(test_dcm_list[:100], total=len(test_dcm_list)))","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:05:08.999077Z","iopub.execute_input":"2023-01-18T21:05:08.999347Z","iopub.status.idle":"2023-01-18T21:05:11.494351Z","shell.execute_reply.started":"2023-01-18T21:05:08.999323Z","shell.execute_reply":"2023-01-18T21:05:11.493217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:05:08.890775Z","iopub.execute_input":"2023-01-18T21:05:08.891062Z","iopub.status.idle":"2023-01-18T21:05:08.925466Z","shell.execute_reply.started":"2023-01-18T21:05:08.891035Z","shell.execute_reply":"2023-01-18T21:05:08.924463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, [256, 256])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:15:10.243092Z","iopub.execute_input":"2023-01-18T21:15:10.243688Z","iopub.status.idle":"2023-01-18T21:15:10.251619Z","shell.execute_reply.started":"2023-01-18T21:15:10.243646Z","shell.execute_reply":"2023-01-18T21:15:10.250479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/working/test_png'\n\ntest_dir = f'{test_path}'\n# /kaggle/working/test_png//10008/736471439.png\n# /kaggle/working/test_png/736471439.png\n\ntest_df['img_path']= f'{test_dir}'\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:19:50.068175Z","iopub.execute_input":"2023-01-18T21:19:50.068555Z","iopub.status.idle":"2023-01-18T21:19:50.087574Z","shell.execute_reply.started":"2023-01-18T21:19:50.068523Z","shell.execute_reply":"2023-01-18T21:19:50.08658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate images\ntest_data = tf.keras.utils.image_dataset_from_directory(\"/kaggle/working/test_png\",\n                                                        labels=None, label_mode=None, color_mode='rgb',\n                                                        image_size=(224,224), shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:21:12.027798Z","iopub.execute_input":"2023-01-18T21:21:12.028213Z","iopub.status.idle":"2023-01-18T21:21:12.145769Z","shell.execute_reply.started":"2023-01-18T21:21:12.028178Z","shell.execute_reply":"2023-01-18T21:21:12.144849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get test prediction\npreds = model.predict_generator(test_data)\n# y_pred = np.argmax(preds, axis=1)\npreds","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:21:54.662729Z","iopub.execute_input":"2023-01-18T21:21:54.663688Z","iopub.status.idle":"2023-01-18T21:21:54.71549Z","shell.execute_reply.started":"2023-01-18T21:21:54.663637Z","shell.execute_reply":"2023-01-18T21:21:54.714516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer = []\nnon_cancer = []\nfor p in preds:\n    cancer.append(p[0])\n    non_cancer.append(p[1])\n\nprint(cancer)\nprint(non_cancer)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:26:17.723274Z","iopub.execute_input":"2023-01-18T21:26:17.723656Z","iopub.status.idle":"2023-01-18T21:26:17.730477Z","shell.execute_reply.started":"2023-01-18T21:26:17.723622Z","shell.execute_reply":"2023-01-18T21:26:17.729514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':cancer, 'non_cancer':non_cancer})\n\npred_df\n# pred_df['cancer']=(pred_df.cancer > 0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:27:35.911206Z","iopub.execute_input":"2023-01-18T21:27:35.911626Z","iopub.status.idle":"2023-01-18T21:27:35.923519Z","shell.execute_reply.started":"2023-01-18T21:27:35.911592Z","shell.execute_reply":"2023-01-18T21:27:35.922508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save the model","metadata":{}},{"cell_type":"code","source":"model_name = 'EffecientNetB3'\nsubject = 'rsna-breast-cancer-detection'\nacc = valid_score[1] * 100\nsave_path = ''\n\nsave_id = str(f'{model_name}-{subject}-{\"%.2f\" %round(acc, 2)}.h5')\nmodel_save_loc = os.path.join(save_path, save_id)\nmodel.save(model_save_loc)\nprint(f'model was saved as {model_save_loc}')","metadata":{"id":"oy5ShUciGG6-","execution":{"iopub.status.busy":"2023-01-18T21:05:13.709838Z","iopub.execute_input":"2023-01-18T21:05:13.710207Z","iopub.status.idle":"2023-01-18T21:05:15.492596Z","shell.execute_reply.started":"2023-01-18T21:05:13.710164Z","shell.execute_reply":"2023-01-18T21:05:15.491513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T21:28:16.440938Z","iopub.execute_input":"2023-01-18T21:28:16.442039Z","iopub.status.idle":"2023-01-18T21:28:16.450743Z","shell.execute_reply.started":"2023-01-18T21:28:16.442001Z","shell.execute_reply":"2023-01-18T21:28:16.449597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}