{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport cv2\nfrom path import Path\nimport os \nimport glob\nimport tensorflow_hub as hub\nimport os \nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport tensorflow as tf\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.utils import to_categorical\nfrom pydicom import dcmread\nimport nibabel as nib\nimport traceback\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-02T10:20:02.345735Z","iopub.execute_input":"2023-04-02T10:20:02.350458Z","iopub.status.idle":"2023-04-02T10:20:08.595729Z","shell.execute_reply.started":"2023-04-02T10:20:02.350406Z","shell.execute_reply":"2023-04-02T10:20:08.594625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/train.csv\")\ntest_df = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/test.csv\")\n\ntrain_dir = '../input/rsna-2022-cervical-spine-fracture-detection/train_images'\ntest_dir = '../input/rsna-2022-cervical-spine-fracture-detection/test_images'\nfirst_image = os.path.join(test_dir, test_df['StudyInstanceUID'].iloc[0])\n\nnew_submission = []\n# means = train_df.median(numeric_only=True).to_dict()\n# means = dict(zip(train_df.columns[1:], np.average(train_df[train_df.columns[1:]], axis=0, weights=train_df[\"patient_overall\"] + 1)))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.597531Z","iopub.execute_input":"2023-04-02T10:20:08.598217Z","iopub.status.idle":"2023-04-02T10:20:08.628925Z","shell.execute_reply.started":"2023-04-02T10:20:08.598177Z","shell.execute_reply":"2023-04-02T10:20:08.628117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction_type = test_df['prediction_type'].tolist()\n# submission = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/sample_submission.csv')\n# for i in range(len(submission)):        \n#     new_submission.append(means[prediction_type[i]])\n# submission['fractured'] = new_submission","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.630356Z","iopub.execute_input":"2023-04-02T10:20:08.630729Z","iopub.status.idle":"2023-04-02T10:20:08.63523Z","shell.execute_reply.started":"2023-04-02T10:20:08.630692Z","shell.execute_reply":"2023-04-02T10:20:08.633986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    try:\n        img=dicom.dcmread(path)\n        img.PhotometricInterpretation = 'YBR_FULL'\n        data=img.pixel_array\n        data=data-np.min(data)\n        \n        if np.max(data) != 0:\n            data=data/np.max(data)\n        data=(data*255).astype(np.uint8)\n#         print(data.shape)\n#         return cv2.cvtColor(data.reshape(512, 512), cv2.COLOR_GRAY2RGB)\n        return cv2.cvtColor(data, cv2.COLOR_GRAY2RGB)\n\n    except Exception:\n        print(f\"image ({path}) not found\")\n        print(traceback.format_exc())\n        return np.zeros((512, 512, 3))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.638505Z","iopub.execute_input":"2023-04-02T10:20:08.639389Z","iopub.status.idle":"2023-04-02T10:20:08.646723Z","shell.execute_reply.started":"2023-04-02T10:20:08.63935Z","shell.execute_reply":"2023-04-02T10:20:08.64582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def listdirs(folder):\n    return [d for d in os.listdir(folder) if os.path.isdir(os.path.join(folder, d))]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.648254Z","iopub.execute_input":"2023-04-02T10:20:08.648938Z","iopub.status.idle":"2023-04-02T10:20:08.657053Z","shell.execute_reply.started":"2023-04-02T10:20:08.648894Z","shell.execute_reply":"2023-04-02T10:20:08.656082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/rsna-2022-cervical-spine-fracture-detection/train_images'\ntest_dir = '../input/rsna-2022-cervical-spine-fracture-detection/test_images'\npatients = sorted(os.listdir(train_dir))\ntest_patients = sorted(os.listdir(test_dir))\npatients[:5]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.65847Z","iopub.execute_input":"2023-04-02T10:20:08.658827Z","iopub.status.idle":"2023-04-02T10:20:08.811645Z","shell.execute_reply.started":"2023-04-02T10:20:08.658788Z","shell.execute_reply":"2023-04-02T10:20:08.810676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_file = glob.glob(\"../input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/*.dcm\")\n# plt.figure(figsize=(20, 20))\n\n# for i in range(28):\n#     ax = plt.subplot(7, 7, i + 1)\n#     image_path = image_file[i]\n#     image = load_dicom(image_path)\n#     plt.axis('off')   \n#     plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.813234Z","iopub.execute_input":"2023-04-02T10:20:08.813595Z","iopub.status.idle":"2023-04-02T10:20:08.81831Z","shell.execute_reply.started":"2023-04-02T10:20:08.813561Z","shell.execute_reply":"2023-04-02T10:20:08.817353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_file = glob.glob(\"../input/rsna-2022-cervical-spine-fracture-detection/segmentations/*.nii\")\n# plt.figure(figsize=(20, 20))\n\n# for i in range(28):\n#     ax = plt.subplot(7, 7, i + 1)\n#     image_path = image_file[i]\n#     nii_img = nib.load(image_path).get_fdata()\n#     nib_image = nii_img[:,:,59]\n#     plt.axis('off')\n#     plt.imshow(nib_image)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.820038Z","iopub.execute_input":"2023-04-02T10:20:08.820725Z","iopub.status.idle":"2023-04-02T10:20:08.826988Z","shell.execute_reply.started":"2023-04-02T10:20:08.820689Z","shell.execute_reply":"2023-04-02T10:20:08.82614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gen_set = []\n\nfor i in range(len(train_df)):\n    \n            idx = train_df.loc[i, 'StudyInstanceUID']\n            labels = list(train_df.iloc[i,-7:])\n            path = os.path.join(train_dir, idx)\n            for filename in os.listdir(path):\n                img_path = os.path.join(path, filename)\n                gen_set.append([idx, img_path] + labels)\n                \ngen_df = pd.DataFrame(gen_set, columns=['ID', 'path'] + list(train_df.columns[-7:]))\ngen_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:20:08.828664Z","iopub.execute_input":"2023-04-02T10:20:08.829306Z","iopub.status.idle":"2023-04-02T10:21:31.047058Z","shell.execute_reply.started":"2023-04-02T10:20:08.829271Z","shell.execute_reply":"2023-04-02T10:21:31.046143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices = list(gen_df.index)\nnp.random.shuffle(indices)\ntrain_idx, val_idx = indices[:int(0.9*len(indices))], indices[int(0.9*len(indices)):]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.05102Z","iopub.execute_input":"2023-04-02T10:21:31.051355Z","iopub.status.idle":"2023-04-02T10:21:31.171542Z","shell.execute_reply.started":"2023-04-02T10:21:31.051327Z","shell.execute_reply":"2023-04-02T10:21:31.17016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen_df = gen_df.iloc[train_idx, :].reset_index(drop=True)\nval_gen_df = gen_df.iloc[val_idx, :].reset_index(drop=True)\n# test_gen_df = gen_df.iloc[test_idx, :].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.172912Z","iopub.execute_input":"2023-04-02T10:21:31.173546Z","iopub.status.idle":"2023-04-02T10:21:31.594826Z","shell.execute_reply.started":"2023-04-02T10:21:31.173504Z","shell.execute_reply":"2023-04-02T10:21:31.593845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_gen_set = []\n\n# for i in range(len(test_df)):\n    \n#     idx = test_df.loc[i, 'StudyInstanceUID']\n#     path = os.path.join(test_dir, idx)\n#     for filename in os.listdir(path):\n#         img_path = os.path.join(path, filename)\n#         test_gen_set.append([idx, img_path])\n                \n# test_gen_df = pd.DataFrame(gen_set, columns=['ID', 'path'])\n# test_gen_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.596337Z","iopub.execute_input":"2023-04-02T10:21:31.5967Z","iopub.status.idle":"2023-04-02T10:21:31.601033Z","shell.execute_reply.started":"2023-04-02T10:21:31.596661Z","shell.execute_reply":"2023-04-02T10:21:31.6Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self, test_gen_df, batch_size=64, input_size=(64, 64), n_channels=3, n_classes=7, shuffle=True):\n        'Initialization'\n        self.gen_df = gen_df\n        self.list_IDs = gen_df['ID']\n        self.batch_size = batch_size\n        self.input_size = input_size\n        self.shuffle = shuffle\n        \n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        \n        self.on_epoch_end()\n        \n    def __get_input(self, path):\n        img = load_dicom(path)\n        img = cv2.resize(img, (64, 64))\n        img_arr = img_to_array(img)\n        img_arr /= 255.0\n        \n        return img_arr\n    \n#     def on_epoch_end(self):\n#         self.indexes = np.arange(len(self.list_IDs))\n#         if self.shuffle == True:\n#             np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        \n        # generate indexes for the current batch\n        ID_indexes = self.indexes[index * self.batch_size : (index+1) * self.batch_size]\n        X = []\n        #generate the list of patient IDs\n        gen_df_temp = gen_df.iloc[ID_indexes, :]\n        list_IDs_temp = gen_df_temp['ID']\n        #generate the data based on gen_df_temp\n        \n        for IDp in list_IDs_temp:\n            idx = gen_df_temp.index[gen_df_temp['ID'] == IDp]\n            path = gen_df_temp.loc[idx, 'path']\n\n            X.append(get_input(path))\n        \n        return np.array(X)\n    \n    def __len__(self):\n        return int(np.floor(self.n // self.batch_size))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.602872Z","iopub.execute_input":"2023-04-02T10:21:31.603582Z","iopub.status.idle":"2023-04-02T10:21:31.615358Z","shell.execute_reply.started":"2023-04-02T10:21:31.603522Z","shell.execute_reply":"2023-04-02T10:21:31.61432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_input(path):\n#     img = load_dicom(path)\n#     img = cv2.resize(img, (64, 64))\n#     img_arr = img_to_array(img)\n#     img_arr /= 255.0\n\n#     return img_arr\n\n\n# list_IDs = gen_df['ID']\n\n# indexes = np.arange(len(list_IDs))\n# np.random.shuffle(indexes)\n\n# ID_indexes = indexes[0 * 64 : (0+1) * 64]\n# # print(ID_indexes)\n# X = []\n# y = []\n# #generate the list of patient IDs\n# gen_df_temp = gen_df.iloc[ID_indexes, :]\n# # print(gen_df_temp)\n# list_IDs_temp = gen_df_temp['ID']\n# #generate the data based on gen_df_temp\n\n# for IDp in list_IDs_temp:\n#     print(IDp)\n#     idx = gen_df_temp.index[gen_df_temp['ID'] == IDp]\n#     path = gen_df_temp.loc[idx, 'path']\n    \n#     print(type(path))\n#     X.append(get_input(path))\n#     y.append(gen_df_temp[gen_df_temp['ID'] == IDp].iloc[:, 2:].values.tolist())\n#     break\n# np.array(X).shape","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.616889Z","iopub.execute_input":"2023-04-02T10:21:31.617309Z","iopub.status.idle":"2023-04-02T10:21:31.626756Z","shell.execute_reply.started":"2023-04-02T10:21:31.617272Z","shell.execute_reply":"2023-04-02T10:21:31.625817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def train_generator(train_df, batch_size, infinite=True, base_path=train_dir):\n#     while True:\n#         train_data = []\n#         train_idx = []\n#         train_label = []\n        \n#         for i in range(len(train_df)):\n#             idx = train_df.loc[i, 'StudyInstanceUID']\n#             path = os.path.join(train_dir, idx)\n#             for filename in os.listdir(path):\n#                 img_path = os.path.join(path, filename)\n# #                 print(img_path)                \n                \n#                 if not \"dcm\" in img_path:\n#                         continue\n                        \n#                 dc  = dicom.read_file(img_path)\n#                 if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#                     continue\n#                 img = load_dicom(img_path)\n#                 img = cv2.resize(img, (64, 64))\n#                 img_arr = img_to_array(img)\n#                 img_arr /= 255.0\n                \n#                 train_data += [img_arr]\n#                 cur_label = []\n#                 cur_label.append(train_df.loc[i, 'patient_overall'])\n#                 cur_label.append(train_df.loc[i,'C1'])\n#                 cur_label.append(train_df.loc[i,'C2'])\n#                 cur_label.append(train_df.loc[i,'C3'])\n#                 cur_label.append(train_df.loc[i,'C4'])\n#                 cur_label.append(train_df.loc[i,'C5'])\n#                 cur_label.append(train_df.loc[i,'C6'])\n#                 cur_label.append(train_df.loc[i,'C7'])\n                \n#                 train_label += [cur_label]\n#                 train_idx += [idx]\n                \n#                 if len(train_idx) == batch_size:\n#                     yield np.array(train_data), np.array(train_label)\n#                     train_data, train_label, train_idx = [], [], []\n                    \n#             i += 1\n            ","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.628175Z","iopub.execute_input":"2023-04-02T10:21:31.62869Z","iopub.status.idle":"2023-04-02T10:21:31.639294Z","shell.execute_reply.started":"2023-04-02T10:21:31.628644Z","shell.execute_reply":"2023-04-02T10:21:31.638337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def test_generator(test_df, infinite = True, base_path = test_dir):\n#     while True:\n#         test_data = []\n#         test_idx = []\n        \n#         for i in range(len(test_df)):\n            \n#             if type(test_df) is list: \n#                 idx = test_df[i]\n#             else:\n#                 idx = test_df.loc[i, 'StudyInstanceUID']\n                \n#             path = os.path.join(test_dir, idx)\n            \n#             if os.path.exists(path):\n#                 for filename in os.listdir(path):\n#                     img_path = os.path.join(path, filename)\n                    \n#                     if not \"dcm\" in img_path:\n#                         continue\n                        \n#                     dc  = dicom.read_file(img_path)\n#                     if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#                         continue\n#                     img = load_dicom(img_path)\n#                     img = cv2.resize(img, (64, 64))\n#                     img_arr = img_to_array(img)\n#                     img_arr /= 255.0\n\n#                     test_data += [img_arr]\n#                     test_idx += [idx]\n\n#                     if len(test_idx) == batch_size:\n#                         yield np.array(test_data)\n#                         train_data = []\n#         print(len(test_data))2\n\n#         if not infinite:\n#             return np.array(test_data)\n#             break\n\n#         if len(test_data) > 0: \n#             yield np.array(test_data)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.64094Z","iopub.execute_input":"2023-04-02T10:21:31.641344Z","iopub.status.idle":"2023-04-02T10:21:31.652321Z","shell.execute_reply.started":"2023-04-02T10:21:31.641291Z","shell.execute_reply":"2023-04-02T10:21:31.651353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import BatchNormalization, Conv2D, MaxPooling2D, Dropout, Flatten, Dense, Activation","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.655431Z","iopub.execute_input":"2023-04-02T10:21:31.655738Z","iopub.status.idle":"2023-04-02T10:21:31.664907Z","shell.execute_reply.started":"2023-04-02T10:21:31.655693Z","shell.execute_reply":"2023-04-02T10:21:31.663955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_spape = (64, 64, 3)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.668057Z","iopub.execute_input":"2023-04-02T10:21:31.668358Z","iopub.status.idle":"2023-04-02T10:21:31.672597Z","shell.execute_reply.started":"2023-04-02T10:21:31.668333Z","shell.execute_reply":"2023-04-02T10:21:31.67165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install efficientnet\n# import efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.674114Z","iopub.execute_input":"2023-04-02T10:21:31.674775Z","iopub.status.idle":"2023-04-02T10:21:31.681343Z","shell.execute_reply.started":"2023-04-02T10:21:31.674737Z","shell.execute_reply":"2023-04-02T10:21:31.680349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_model():\n#     tf.keras.applications.efficientnet.EfficientNetB7(input_shape=input_spape,weights='imagenet',include_top=False),\n#     tf.keras.layers.GlobalAveragePooling2D(),\n#     tf.keras.layers.Dropout(0.25),\n#     tf.keras.layers.Dense(256, activation='swish'),\n#     tf.keras.layers.Dropout(0.25),\n#     tf.keras.layers.Dense(7, activation='sigmoid')\n    \n    \n#     model.compile(loss=\"binary_crossentropy\", \n#                   optimizer = tf.keras.optimizers.Nadam(learning_rate = 0.001),\n#                   metrics=[tf.keras.metrics.BinaryAccuracy(), tf.keras.metrics.AUC(multi_label=True)])\n#     model.summary()\n    \n#     return model","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.682932Z","iopub.execute_input":"2023-04-02T10:21:31.683711Z","iopub.status.idle":"2023-04-02T10:21:31.690001Z","shell.execute_reply.started":"2023-04-02T10:21:31.683552Z","shell.execute_reply":"2023-04-02T10:21:31.689134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(32, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.2))\n\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(64, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Dropout(0.2))\n\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(128, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25))\n    \n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(256, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25))\n    \n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(256, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25)) \n    \n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(128, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25))   \n\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(64, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.2))\n\n    model.add(Flatten())\n    model.add(Dense(1024))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(128))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(32))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(7))\n    model.add(Activation('softmax'))\n    \n    model.summary()\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Nadam(learning_rate = 0.001),\n                  metrics=[tf.keras.metrics.BinaryAccuracy(), tf.keras.metrics.AUC(multi_label=True)])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.691855Z","iopub.execute_input":"2023-04-02T10:21:31.692605Z","iopub.status.idle":"2023-04-02T10:21:31.741518Z","shell.execute_reply.started":"2023-04-02T10:21:31.692568Z","shell.execute_reply":"2023-04-02T10:21:31.74039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self, gen_df, batch_size=64, input_size=(64, 64), n_channels=3, n_classes=7, shuffle=True):\n        'Initialization'\n        self.gen_df = gen_df\n        self.list_IDs = gen_df['ID']\n        self.batch_size = batch_size\n        self.input_size = input_size\n        self.shuffle = shuffle\n        \n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        \n        self.on_epoch_end()\n        \n    def __get_input(self, path):\n        img = load_dicom(path)\n        img = cv2.resize(img, (64, 64))\n        img_arr = img_to_array(img)\n        img_arr /= 255.0\n        \n        return img_arr\n    \n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        \n        # generate indexes for the current batch\n        ID_indexes = self.indexes[index * self.batch_size : (index+1) * self.batch_size]\n        X = []\n        y = []\n        #generate the list of patient IDs\n        gen_df_temp = gen_df.iloc[ID_indexes, :]\n        #generate the data based on gen_df_temp\n        \n        for i in range(len(gen_df_temp)):\n            path = gen_df_temp.iloc[i, 1]\n            dc = dicom.read_file(path)\n            if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n                continue\n                \n            X.append(self.__get_input(path))\n            y.append(gen_df_temp.iloc[i, 2:].values.tolist())\n        \n        return np.array(X), np.array(y)\n    \n    def __len__(self):\n        return int(np.floor(len(self.list_IDs) // self.batch_size))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.743379Z","iopub.execute_input":"2023-04-02T10:21:31.74418Z","iopub.status.idle":"2023-04-02T10:21:31.756574Z","shell.execute_reply.started":"2023-04-02T10:21:31.744141Z","shell.execute_reply":"2023-04-02T10:21:31.755583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = TrainDataGen(train_gen_df)\nval_generator = TrainDataGen(val_gen_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:12:39.626464Z","iopub.execute_input":"2023-04-02T11:12:39.626823Z","iopub.status.idle":"2023-04-02T11:12:39.650037Z","shell.execute_reply.started":"2023-04-02T11:12:39.626792Z","shell.execute_reply":"2023-04-02T11:12:39.649065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tra","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_generator[10][1])","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:17:34.641465Z","iopub.execute_input":"2023-04-02T11:17:34.642031Z","iopub.status.idle":"2023-04-02T11:17:35.837144Z","shell.execute_reply.started":"2023-04-02T11:17:34.641982Z","shell.execute_reply":"2023-04-02T11:17:35.836071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"val_generator= np.array(val_generator)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:19:38.256369Z","iopub.execute_input":"2023-04-02T11:19:38.256737Z","iopub.status.idle":"2023-04-02T11:33:16.911421Z","shell.execute_reply.started":"2023-04-02T11:19:38.256706Z","shell.execute_reply":"2023-04-02T11:33:16.909271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:45:49.990076Z","iopub.execute_input":"2023-04-02T11:45:49.990518Z","iopub.status.idle":"2023-04-02T11:45:50.063287Z","shell.execute_reply.started":"2023-04-02T11:45:49.990483Z","shell.execute_reply":"2023-04-02T11:45:50.062305Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test1=[]\ny_test=[]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T12:16:05.417698Z","iopub.execute_input":"2023-04-02T12:16:05.418023Z","iopub.status.idle":"2023-04-02T12:16:05.423186Z","shell.execute_reply.started":"2023-04-02T12:16:05.41799Z","shell.execute_reply":"2023-04-02T12:16:05.422244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(val_generator)):\n    y_test1.append(val_generator[i][1])","metadata":{"execution":{"iopub.status.busy":"2023-04-02T12:16:09.827742Z","iopub.execute_input":"2023-04-02T12:16:09.828132Z","iopub.status.idle":"2023-04-02T12:16:09.833557Z","shell.execute_reply.started":"2023-04-02T12:16:09.828077Z","shell.execute_reply":"2023-04-02T12:16:09.832583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in y_test1:\n    for j in i:\n        y_test.append(j)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T12:16:13.76607Z","iopub.execute_input":"2023-04-02T12:16:13.767297Z","iopub.status.idle":"2023-04-02T12:16:13.791247Z","shell.execute_reply.started":"2023-04-02T12:16:13.767229Z","shell.execute_reply":"2023-04-02T12:16:13.790082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T12:17:34.036861Z","iopub.execute_input":"2023-04-02T12:17:34.037256Z","iopub.status.idle":"2023-04-02T12:17:34.04434Z","shell.execute_reply.started":"2023-04-02T12:17:34.037221Z","shell.execute_reply":"2023-04-02T12:17:34.043251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n# tf.debugging.set_log_device_placement(False)\nx_train = train_df.reset_index()\n# x_val = train_df.iloc[val_idx].reset_index()\n\nmodel = create_model()\n# steps_per_epoch = int(len(train_gen_df) / 64)\nhist = model.fit(train_generator,\n                    epochs = 100,\n                    verbose = 1,\n                    callbacks = [tf.keras.callbacks.EarlyStopping(monitor=\"loss\", min_delta=0.005, \n                         patience=15)],\n    #                             validation_steps = max((len(x_val) // 64), 1),\n                    steps_per_epoch = 25,\n#                     validation_data = val_generator,\n#                     val_pred = model.predict(val_generator),\n#                                       steps = max((len(test_df) // 64), 1)\n                )\n\ntry: # the best we can do at the moment..\n    preds = model.predict_generator(test_generator(test_df, min(len(test_df), 64), infinite = False, base_path = test_dir), steps = max((len(test_df) // 64), 1))\n\n    new_preds = []\n    for pred_idx in range(len(preds)):\n        new_preds.append(preds[pred_idx][prediction_type_mapping[pred_idx]])\n    # submission['fractured'] += preds[:, prediction_type_mapping] / 5\n    submission['fractured'] += np.array(new_preds) / 5\n\nexcept: traceback.print_exc()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:21:31.806986Z","iopub.execute_input":"2023-04-02T10:21:31.80805Z","iopub.status.idle":"2023-04-02T10:44:08.258058Z","shell.execute_reply.started":"2023-04-02T10:21:31.808011Z","shell.execute_reply":"2023-04-02T10:44:08.256012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_input(path):\n#     img = load_dicom(path)\n#     img = cv2.resize(img, (64, 64))\n#     img_arr = img_to_array(img)\n#     img_arr /= 255.0\n\n#     return img_arr\n\n# def get_val_data(val_gen_df):\n#     X = []\n#     y = []\n#     #generate the list of patient IDs\n#     gen_df_temp = val_gen_df\n#     #generate the data based on gen_df_temp\n\n#     for i in range(len(gen_df_temp)):\n#         path = gen_df_temp.iloc[i, 1]\n#         dc = dicom.read_file(path)\n#         if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#             continue\n\n#         X.append(get_input(path))\n#         y.append(gen_df_temp.iloc[i, 2:].values.tolist())\n\n#     return np.array(X), np.array(y)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.262885Z","iopub.execute_input":"2023-04-02T10:44:08.264934Z","iopub.status.idle":"2023-04-02T10:44:08.273259Z","shell.execute_reply.started":"2023-04-02T10:44:08.264892Z","shell.execute_reply":"2023-04-02T10:44:08.27204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_gen_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.283651Z","iopub.execute_input":"2023-04-02T10:44:08.286137Z","iopub.status.idle":"2023-04-02T10:44:08.306949Z","shell.execute_reply.started":"2023-04-02T10:44:08.286083Z","shell.execute_reply":"2023-04-02T10:44:08.305667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_X, val_y = get_val_data(val_gen_df.iloc[:1000, :])","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.310389Z","iopub.execute_input":"2023-04-02T10:44:08.311303Z","iopub.status.idle":"2023-04-02T10:44:08.318466Z","shell.execute_reply.started":"2023-04-02T10:44:08.311263Z","shell.execute_reply":"2023-04-02T10:44:08.317165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred = hist.model.predict(val_generator)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:45:25.053618Z","iopub.execute_input":"2023-04-02T10:45:25.053976Z","iopub.status.idle":"2023-04-02T11:03:38.439853Z","shell.execute_reply.started":"2023-04-02T10:45:25.053944Z","shell.execute_reply":"2023-04-02T11:03:38.435745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.333Z","iopub.execute_input":"2023-04-02T10:44:08.334312Z","iopub.status.idle":"2023-04-02T10:44:08.342066Z","shell.execute_reply.started":"2023-04-02T10:44:08.334273Z","shell.execute_reply":"2023-04-02T10:44:08.341012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred[val_pred >= 0.22] = 1\nval_pred[val_pred < 0.22] = 0","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:05:25.910057Z","iopub.execute_input":"2023-04-02T11:05:25.911059Z","iopub.status.idle":"2023-04-02T11:05:25.918888Z","shell.execute_reply.started":"2023-04-02T11:05:25.911005Z","shell.execute_reply":"2023-04-02T11:05:25.917889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for train_idx, val_idx in StratifiedKFold(5).split(train_df, train_df['patient_overall']):\n#     tf.keras.backend.clear_session()\n#     x_train = train_df.iloc[train_idx].reset_index()\n#     x_val = train_df.iloc[val_idx].reset_index()\n    \n#     model = create_model()\n    \n#     hist = model.fit(train_generator(x_train, min(len(x_train), 64), infinite=False, base_path=train_dir),\n#                                 epochs = 50,\n#                                 verbose = 1,\n# #                                 callbacks = [tf.keras.callbacks.EarlyStopping(monitor=\"val_loss\", min_delta=0.001, \n# #                                      patience=10)],\n#                                 validation_steps = max((len(x_val) // 64), 1),\n#                                 steps_per_epoch = max((len(x_train) // 64), 1),\n#                                 validation_data = train_generator(x_val, min(len(x_val), 64), infinite = False, base_path = train_dir))\n#     val_pred = model.predict(train_generator(x_val, min(len(test_df), 64), infinite = False, base_path = train_dir), \n#                                       steps = max((len(test_df) // 64), 1))\n    \n#     try: # the best we can do at the moment..\n#         preds = model.predict_generator(test_generator(test_df, min(len(test_df), 64), infinite = False, base_path = test_dir), steps = max((len(test_df) // 64), 1))\n        \n#         new_preds = []\n#         for pred_idx in range(len(preds)):\n#             new_preds.append(preds[pred_idx][prediction_type_mapping[pred_idx]])\n#         # submission['fractured'] += preds[:, prediction_type_mapping] / 5\n#         submission['fractured'] += np.array(new_preds) / 5\n        \n#     except: traceback.print_exc()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.355913Z","iopub.execute_input":"2023-04-02T10:44:08.356341Z","iopub.status.idle":"2023-04-02T10:44:08.364669Z","shell.execute_reply.started":"2023-04-02T10:44:08.356312Z","shell.execute_reply":"2023-04-02T10:44:08.363498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_dir='../input/rsna-2022-cervical-spine-fracture-detection/test_images'\n# testset=[]\n# testidt=[]\n# for i in tqdm(range(len(test_df))):\n#     idt=test_df[i]\n#     path=os.path.join(test_dir,idt)   \n    \n#     for im in os.listdir(path):\n#         dc = dicom.read_file(os.path.join(path,im))\n        \n#         if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#             continue\n#         img=load_dicom(os.path.join(path,im)) \n\n#         img=cv.resize(img,(64,64)) \n#         image=img_to_array(img)\n#         image=image/255.0\n#         testset+=[image]\n#         testidt+=[idt]\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.366457Z","iopub.execute_input":"2023-04-02T10:44:08.367413Z","iopub.status.idle":"2023-04-02T10:44:08.381779Z","shell.execute_reply.started":"2023-04-02T10:44:08.367363Z","shell.execute_reply":"2023-04-02T10:44:08.380357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist.model.save('model_01')","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:08.383814Z","iopub.execute_input":"2023-04-02T10:44:08.384547Z","iopub.status.idle":"2023-04-02T10:44:13.540048Z","shell.execute_reply.started":"2023-04-02T10:44:08.384505Z","shell.execute_reply":"2023-04-02T10:44:13.539081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.plot(hist.history['loss'])\n# plt.plot(hist.history['binary_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('binary_accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.542008Z","iopub.execute_input":"2023-04-02T10:44:13.543166Z","iopub.status.idle":"2023-04-02T10:44:13.857841Z","shell.execute_reply.started":"2023-04-02T10:44:13.543125Z","shell.execute_reply":"2023-04-02T10:44:13.848756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! ls /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.874379Z","iopub.execute_input":"2023-04-02T10:44:13.882442Z","iopub.status.idle":"2023-04-02T10:44:13.904038Z","shell.execute_reply.started":"2023-04-02T10:44:13.882372Z","shell.execute_reply":"2023-04-02T10:44:13.901962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a href=\"./model_01\"> Download File </a>","metadata":{}},{"cell_type":"code","source":"# test_data = []\n# test_idx = []\n\n# for i in range(len(test_df)):\n\n#     if type(test_df) is list: \n#         idx = test_df[i]\n#     else:\n#         idx = test_df.loc[i, 'StudyInstanceUID']\n\n#     path = os.path.join(test_dir, idx)\n#     print(path)\n#     if os.path.exists(path):\n#         for filename in os.listdir(path):\n#             img_path = os.path.join(path, filename)\n\n#             if not \"dcm\" in img_path:\n#                 continue\n\n#             dc  = dicom.read_file(img_path)\n#             if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#                 continue\n#             img = load_dicom(img_path)\n#             img = cv2.resize(img, (64, 64))\n#             img_arr = img_to_array(img)\n#             img_arr /= 255.0\n\n#             test_data += [img_arr]\n#             test_idx += [idx]\n            \n#     else:\n#         print(\"problem with the path\")\n            \n# print(len(test_data))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.90873Z","iopub.execute_input":"2023-04-02T10:44:13.912434Z","iopub.status.idle":"2023-04-02T10:44:13.92274Z","shell.execute_reply.started":"2023-04-02T10:44:13.912387Z","shell.execute_reply":"2023-04-02T10:44:13.921131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds = hist.model.predict(test_generator(test_df, min(len(test_df), 64), infinite = False, base_path = test_dir), steps = max((len(test_df) // 64)))\n\n# new_preds = []\n# for pred_idx in range(len(preds)):\n#     new_preds.append(preds[pred_idx][prediction_type_mapping[pred_idx]])\n# # submission['fractured'] += preds[:, prediction_type_mapping] / 5\n# submission['fractured'] += np.array(new_preds) / 5","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.925008Z","iopub.execute_input":"2023-04-02T10:44:13.92754Z","iopub.status.idle":"2023-04-02T10:44:13.93586Z","shell.execute_reply.started":"2023-04-02T10:44:13.927502Z","shell.execute_reply":"2023-04-02T10:44:13.934333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tf.keras.backend.clear_session()\n\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# # instantiate a distribution strategy\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\n# print(\"All devices: \", tf.config.list_logical_devices('TPU'))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.942899Z","iopub.execute_input":"2023-04-02T10:44:13.945021Z","iopub.status.idle":"2023-04-02T10:44:13.954026Z","shell.execute_reply.started":"2023-04-02T10:44:13.944973Z","shell.execute_reply":"2023-04-02T10:44:13.952441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tf.keras.utils.plot_model(model, show_shapes=True, rankdir='TB')","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.956541Z","iopub.execute_input":"2023-04-02T10:44:13.957643Z","iopub.status.idle":"2023-04-02T10:44:13.969305Z","shell.execute_reply.started":"2023-04-02T10:44:13.957574Z","shell.execute_reply":"2023-04-02T10:44:13.968231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = create_model()\n# model.compile(\n#   optimizer=tf.keras.optimizers.Nadam(learning_rate=0.0005),\n#   loss='binary_crossentropy',\n#   metrics=['accuracy'])\n\n# callbacks = [\n#     # TensorBoard will store logs for each epoch and graph performance for us.\n#     # tf.keras.callbacks.TensorBoard(log_dir=log_dir, histogram_freq=1),\n#     # ModelCheckpoint will save models after each epoch for retrieval later.\n#     # tf.keras.callbacks.ModelCheckpoint(checkpoint_path),\n#     # EarlyStopping will terminate training when val_loss ceases to improve.\n#     tf.keras.callbacks.EarlyStopping(monitor=\"loss\", min_delta=0.001, \n#                                      patience=10, restore_best_weights=True)\n# ]\n\n\n# model.fit(\n#     X_train, Y_train,\n#     epochs=100,\n#     batch_size=64,\n#     callbacks=callbacks\n# )\n\n# model.save_weights('./fashion_mnist.h5', overwrite=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.970736Z","iopub.execute_input":"2023-04-02T10:44:13.972351Z","iopub.status.idle":"2023-04-02T10:44:13.981679Z","shell.execute_reply.started":"2023-04-02T10:44:13.972309Z","shell.execute_reply":"2023-04-02T10:44:13.980647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred=model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.984614Z","iopub.execute_input":"2023-04-02T10:44:13.985653Z","iopub.status.idle":"2023-04-02T10:44:13.996965Z","shell.execute_reply.started":"2023-04-02T10:44:13.985613Z","shell.execute_reply":"2023-04-02T10:44:13.994661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result = pd.DataFrame(columns = train_df.columns, index = range(len(testidt)))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:13.99908Z","iopub.execute_input":"2023-04-02T10:44:14.000009Z","iopub.status.idle":"2023-04-02T10:44:14.00634Z","shell.execute_reply.started":"2023-04-02T10:44:13.999964Z","shell.execute_reply":"2023-04-02T10:44:14.00515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in tqdm(range(len(testidt))):\n#     result.loc[i, 'StudyInstanceUID'] = testidt[i]\n#     rows = np.int64(y_pred[i]>0.07)\n#     result.loc[i, 'patient_overall'] = int(bool(np.sum(rows)))\n#     result.loc[i, 'C1'] = rows[0]\n#     result.loc[i, 'C2'] = rows[1]\n#     result.loc[i, 'C3'] = rows[2]\n#     result.loc[i, 'C4'] = rows[3]\n#     result.loc[i, 'C5'] = rows[4]\n#     result.loc[i, 'C6'] = rows[5]\n#     result.loc[i, 'C7'] = rows[6]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.009448Z","iopub.execute_input":"2023-04-02T10:44:14.011359Z","iopub.status.idle":"2023-04-02T10:44:14.021145Z","shell.execute_reply.started":"2023-04-02T10:44:14.01132Z","shell.execute_reply":"2023-04-02T10:44:14.020268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# means = result[['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']].mean().to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.022686Z","iopub.execute_input":"2023-04-02T10:44:14.02355Z","iopub.status.idle":"2023-04-02T10:44:14.036495Z","shell.execute_reply.started":"2023-04-02T10:44:14.023502Z","shell.execute_reply":"2023-04-02T10:44:14.035356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df2 = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/test.csv')\n# # means = train_df.mean(numeric_only=True).to_dict()\n# test_df2['fractured'] = test_df2['prediction_type'].map(means)\n\n# test_df2[['row_id','fractured']].to_csv('submission.csv', index=False, float_format='%.1g')","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.037753Z","iopub.execute_input":"2023-04-02T10:44:14.038145Z","iopub.status.idle":"2023-04-02T10:44:14.046848Z","shell.execute_reply.started":"2023-04-02T10:44:14.038081Z","shell.execute_reply":"2023-04-02T10:44:14.045832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.047977Z","iopub.execute_input":"2023-04-02T10:44:14.049047Z","iopub.status.idle":"2023-04-02T10:44:14.055331Z","shell.execute_reply.started":"2023-04-02T10:44:14.04901Z","shell.execute_reply":"2023-04-02T10:44:14.054235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plotConfusionMatrix(y_true, y_pred):\n    conf_matrix = confusion_matrix(y_true, y_pred)\n    norm_array = conf_matrix.astype('float') / conf_matrix.sum(axis=1)[:, np.newaxis]\n    group_counts = [\"{0:0.0f}\".format(value) for value in conf_matrix.flatten()]\n    group_percentages = [\"{0:.2%}\".format(value) for value in norm_array.flatten()]\n    labels = [f\"{v1}\\n\\n{v2}\" for v1, v2 in zip(group_counts, group_percentages)]\n    labels = np.asarray(labels).reshape(2, 2)\n    df_cm = pd.DataFrame(conf_matrix, range(2), range(2))\n    ax = sn.heatmap(df_cm, annot=labels, fmt='', cmap='Greens')\n    ax.set_title('CNN model confusion matrix');\n    ax.set_xlabel('Predicted Values')\n    ax.set_ylabel('Actual Values');\n    ## Ticket labels - List must be in alphabetical order\n    ax.xaxis.set_ticklabels(['Non-Likely-Fracture', 'Likely-a-Fracture'])\n    ax.yaxis.set_ticklabels(['Non-Likely-Fracture', 'Likely-a-Fracture'], va=\"center\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.057239Z","iopub.execute_input":"2023-04-02T10:44:14.057971Z","iopub.status.idle":"2023-04-02T10:44:14.077162Z","shell.execute_reply.started":"2023-04-02T10:44:14.057934Z","shell.execute_reply":"2023-04-02T10:44:14.075539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plotConfusionMatix()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.078723Z","iopub.status.idle":"2023-04-02T10:44:14.08722Z","shell.execute_reply.started":"2023-04-02T10:44:14.079915Z","shell.execute_reply":"2023-04-02T10:44:14.079941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:05:37.311142Z","iopub.execute_input":"2023-04-02T11:05:37.311556Z","iopub.status.idle":"2023-04-02T11:05:37.320026Z","shell.execute_reply.started":"2023-04-02T11:05:37.31152Z","shell.execute_reply":"2023-04-02T11:05:37.318661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator[0][1]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T11:07:01.732914Z","iopub.execute_input":"2023-04-02T11:07:01.733399Z","iopub.status.idle":"2023-04-02T11:07:02.282487Z","shell.execute_reply.started":"2023-04-02T11:07:01.733359Z","shell.execute_reply":"2023-04-02T11:07:02.281302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, val_pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T12:17:55.098731Z","iopub.execute_input":"2023-04-02T12:17:55.099138Z","iopub.status.idle":"2023-04-02T12:17:55.129773Z","shell.execute_reply.started":"2023-04-02T12:17:55.09908Z","shell.execute_reply":"2023-04-02T12:17:55.128123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_df = pd.DataFrame(cm,\n                     index = ['C1','C2','C3','C4','C5','C6','C7'], \n                     columns = ['C1','C2','C3','C4','C5','C6','C7'])","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.091187Z","iopub.status.idle":"2023-04-02T10:44:14.091963Z","shell.execute_reply.started":"2023-04-02T10:44:14.091713Z","shell.execute_reply":"2023-04-02T10:44:14.091738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5,4))\nsns.heatmap(cm_df, annot=True)\nplt.title('Confusion Matrix')\nplt.ylabel('Actal Values')\nplt.xlabel('Predicted Values')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T10:44:14.093421Z","iopub.status.idle":"2023-04-02T10:44:14.094212Z","shell.execute_reply.started":"2023-04-02T10:44:14.093933Z","shell.execute_reply":"2023-04-02T10:44:14.093958Z"},"trusted":true},"execution_count":null,"outputs":[]}]}