{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport cv2\nfrom path import Path\nimport os \nimport glob\nimport tensorflow_hub as hub\nimport os \nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport tensorflow as tf\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.utils import to_categorical\nfrom pydicom import dcmread\nimport nibabel as nib\nimport traceback\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-23T21:09:56.293495Z","iopub.execute_input":"2022-08-23T21:09:56.294008Z","iopub.status.idle":"2022-08-23T21:09:56.307961Z","shell.execute_reply.started":"2022-08-23T21:09:56.293965Z","shell.execute_reply":"2022-08-23T21:09:56.306618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/train.csv\")\ntest_df = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/test.csv\")\n\ntrain_dir = '../input/rsna-2022-cervical-spine-fracture-detection/train_images'\ntest_dir = '../input/rsna-2022-cervical-spine-fracture-detection/test_images'\nfirst_image = os.path.join(test_dir, test_df['StudyInstanceUID'].iloc[0])\n\nnew_submission = []\n# means = train_df.median(numeric_only=True).to_dict()\n# means = dict(zip(train_df.columns[1:], np.average(train_df[train_df.columns[1:]], axis=0, weights=train_df[\"patient_overall\"] + 1)))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:56.778556Z","iopub.execute_input":"2022-08-23T21:09:56.779453Z","iopub.status.idle":"2022-08-23T21:09:56.815899Z","shell.execute_reply.started":"2022-08-23T21:09:56.779408Z","shell.execute_reply":"2022-08-23T21:09:56.814772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction_type = test_df['prediction_type'].tolist()\n# submission = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/sample_submission.csv')\n# for i in range(len(submission)):        \n#     new_submission.append(means[prediction_type[i]])\n# submission['fractured'] = new_submission","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:56.984631Z","iopub.execute_input":"2022-08-23T21:09:56.98516Z","iopub.status.idle":"2022-08-23T21:09:56.990566Z","shell.execute_reply.started":"2022-08-23T21:09:56.985117Z","shell.execute_reply":"2022-08-23T21:09:56.989477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    try:\n        img=dicom.dcmread(path)\n        img.PhotometricInterpretation = 'YBR_FULL'\n        data=img.pixel_array\n        data=data-np.min(data)\n        \n        if np.max(data) != 0:\n            data=data/np.max(data)\n        data=(data*255).astype(np.uint8)\n#         print(data.shape)\n#         return cv2.cvtColor(data.reshape(512, 512), cv2.COLOR_GRAY2RGB)\n        return cv2.cvtColor(data, cv2.COLOR_GRAY2RGB)\n\n    except Exception:\n        print(f\"image ({path}) not found\")\n        print(traceback.format_exc())\n        return np.zeros((512, 512, 3))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:57.206435Z","iopub.execute_input":"2022-08-23T21:09:57.20707Z","iopub.status.idle":"2022-08-23T21:09:57.232332Z","shell.execute_reply.started":"2022-08-23T21:09:57.207016Z","shell.execute_reply":"2022-08-23T21:09:57.230876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def listdirs(folder):\n    return [d for d in os.listdir(folder) if os.path.isdir(os.path.join(folder, d))]","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:57.438279Z","iopub.execute_input":"2022-08-23T21:09:57.439534Z","iopub.status.idle":"2022-08-23T21:09:57.445581Z","shell.execute_reply.started":"2022-08-23T21:09:57.439489Z","shell.execute_reply":"2022-08-23T21:09:57.444257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/rsna-2022-cervical-spine-fracture-detection/train_images'\ntest_dir = '../input/rsna-2022-cervical-spine-fracture-detection/test_images'\npatients = sorted(os.listdir(train_dir))\ntest_patients = sorted(os.listdir(test_dir))\npatients[:5]","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:57.904917Z","iopub.execute_input":"2022-08-23T21:09:57.905495Z","iopub.status.idle":"2022-08-23T21:09:58.033917Z","shell.execute_reply.started":"2022-08-23T21:09:57.905443Z","shell.execute_reply":"2022-08-23T21:09:58.032668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_file = glob.glob(\"../input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/*.dcm\")\n# plt.figure(figsize=(20, 20))\n\n# for i in range(28):\n#     ax = plt.subplot(7, 7, i + 1)\n#     image_path = image_file[i]\n#     image = load_dicom(image_path)\n#     plt.axis('off')   \n#     plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:58.423643Z","iopub.execute_input":"2022-08-23T21:09:58.424153Z","iopub.status.idle":"2022-08-23T21:09:58.432956Z","shell.execute_reply.started":"2022-08-23T21:09:58.424107Z","shell.execute_reply":"2022-08-23T21:09:58.431724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_file = glob.glob(\"../input/rsna-2022-cervical-spine-fracture-detection/segmentations/*.nii\")\n# plt.figure(figsize=(20, 20))\n\n# for i in range(28):\n#     ax = plt.subplot(7, 7, i + 1)\n#     image_path = image_file[i]\n#     nii_img = nib.load(image_path).get_fdata()\n#     nib_image = nii_img[:,:,59]\n#     plt.axis('off')\n#     plt.imshow(nib_image)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:58.703867Z","iopub.execute_input":"2022-08-23T21:09:58.704496Z","iopub.status.idle":"2022-08-23T21:09:58.709635Z","shell.execute_reply.started":"2022-08-23T21:09:58.70446Z","shell.execute_reply":"2022-08-23T21:09:58.708398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gen_set = []\n\nfor i in range(len(train_df)):\n    \n            idx = train_df.loc[i, 'StudyInstanceUID']\n            labels = list(train_df.iloc[i,-7:])\n            path = os.path.join(train_dir, idx)\n            for filename in os.listdir(path):\n                img_path = os.path.join(path, filename)\n                gen_set.append([idx, img_path] + labels)\n                \ngen_df = pd.DataFrame(gen_set, columns=['ID', 'path'] + list(train_df.columns[-7:]))\ngen_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:09:58.920302Z","iopub.execute_input":"2022-08-23T21:09:58.92149Z","iopub.status.idle":"2022-08-23T21:11:01.223786Z","shell.execute_reply.started":"2022-08-23T21:09:58.921443Z","shell.execute_reply":"2022-08-23T21:11:01.222658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices = list(gen_df.index)\nnp.random.shuffle(indices)\ntrain_idx, val_idx = indices[:int(0.9*len(indices))], indices[int(0.9*len(indices)):]","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:01.225779Z","iopub.execute_input":"2022-08-23T21:11:01.226375Z","iopub.status.idle":"2022-08-23T21:11:01.41079Z","shell.execute_reply.started":"2022-08-23T21:11:01.226335Z","shell.execute_reply":"2022-08-23T21:11:01.409731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen_df = gen_df.iloc[train_idx, :].reset_index(drop=True)\nval_gen_df = gen_df.iloc[val_idx, :].reset_index(drop=True)\n# test_gen_df = gen_df.iloc[test_idx, :].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:01.412224Z","iopub.execute_input":"2022-08-23T21:11:01.413283Z","iopub.status.idle":"2022-08-23T21:11:02.330422Z","shell.execute_reply.started":"2022-08-23T21:11:01.413244Z","shell.execute_reply":"2022-08-23T21:11:02.329377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_gen_set = []\n\n# for i in range(len(test_df)):\n    \n#     idx = test_df.loc[i, 'StudyInstanceUID']\n#     path = os.path.join(test_dir, idx)\n#     for filename in os.listdir(path):\n#         img_path = os.path.join(path, filename)\n#         test_gen_set.append([idx, img_path])\n                \n# test_gen_df = pd.DataFrame(gen_set, columns=['ID', 'path'])\n# test_gen_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.416039Z","iopub.execute_input":"2022-08-23T21:11:02.416794Z","iopub.status.idle":"2022-08-23T21:11:02.421877Z","shell.execute_reply.started":"2022-08-23T21:11:02.416751Z","shell.execute_reply":"2022-08-23T21:11:02.420757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self, test_gen_df, batch_size=64, input_size=(64, 64), n_channels=3, n_classes=7, shuffle=True):\n        'Initialization'\n        self.gen_df = gen_df\n        self.list_IDs = gen_df['ID']\n        self.batch_size = batch_size\n        self.input_size = input_size\n        self.shuffle = shuffle\n        \n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        \n        self.on_epoch_end()\n        \n    def __get_input(self, path):\n        img = load_dicom(path)\n        img = cv2.resize(img, (64, 64))\n        img_arr = img_to_array(img)\n        img_arr /= 255.0\n        \n        return img_arr\n    \n#     def on_epoch_end(self):\n#         self.indexes = np.arange(len(self.list_IDs))\n#         if self.shuffle == True:\n#             np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        \n        # generate indexes for the current batch\n        ID_indexes = self.indexes[index * self.batch_size : (index+1) * self.batch_size]\n        X = []\n        #generate the list of patient IDs\n        gen_df_temp = gen_df.iloc[ID_indexes, :]\n        list_IDs_temp = gen_df_temp['ID']\n        #generate the data based on gen_df_temp\n        \n        for IDp in list_IDs_temp:\n            idx = gen_df_temp.index[gen_df_temp['ID'] == IDp]\n            path = gen_df_temp.loc[idx, 'path']\n\n            X.append(get_input(path))\n        \n        return np.array(X)\n    \n    def __len__(self):\n        return int(np.floor(self.n // self.batch_size))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.42378Z","iopub.execute_input":"2022-08-23T21:11:02.424479Z","iopub.status.idle":"2022-08-23T21:11:02.43601Z","shell.execute_reply.started":"2022-08-23T21:11:02.42444Z","shell.execute_reply":"2022-08-23T21:11:02.435017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_input(path):\n#     img = load_dicom(path)\n#     img = cv2.resize(img, (64, 64))\n#     img_arr = img_to_array(img)\n#     img_arr /= 255.0\n\n#     return img_arr\n\n\n# list_IDs = gen_df['ID']\n\n# indexes = np.arange(len(list_IDs))\n# np.random.shuffle(indexes)\n\n# ID_indexes = indexes[0 * 64 : (0+1) * 64]\n# # print(ID_indexes)\n# X = []\n# y = []\n# #generate the list of patient IDs\n# gen_df_temp = gen_df.iloc[ID_indexes, :]\n# # print(gen_df_temp)\n# list_IDs_temp = gen_df_temp['ID']\n# #generate the data based on gen_df_temp\n\n# for IDp in list_IDs_temp:\n#     print(IDp)\n#     idx = gen_df_temp.index[gen_df_temp['ID'] == IDp]\n#     path = gen_df_temp.loc[idx, 'path']\n    \n#     print(type(path))\n#     X.append(get_input(path))\n#     y.append(gen_df_temp[gen_df_temp['ID'] == IDp].iloc[:, 2:].values.tolist())\n#     break\n# np.array(X).shape","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:15:46.017703Z","iopub.execute_input":"2022-08-23T20:15:46.018156Z","iopub.status.idle":"2022-08-23T20:15:46.029758Z","shell.execute_reply.started":"2022-08-23T20:15:46.018118Z","shell.execute_reply":"2022-08-23T20:15:46.028705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def train_generator(train_df, batch_size, infinite=True, base_path=train_dir):\n#     while True:\n#         train_data = []\n#         train_idx = []\n#         train_label = []\n        \n#         for i in range(len(train_df)):\n#             idx = train_df.loc[i, 'StudyInstanceUID']\n#             path = os.path.join(train_dir, idx)\n#             for filename in os.listdir(path):\n#                 img_path = os.path.join(path, filename)\n# #                 print(img_path)                \n                \n#                 if not \"dcm\" in img_path:\n#                         continue\n                        \n#                 dc  = dicom.read_file(img_path)\n#                 if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#                     continue\n#                 img = load_dicom(img_path)\n#                 img = cv2.resize(img, (64, 64))\n#                 img_arr = img_to_array(img)\n#                 img_arr /= 255.0\n                \n#                 train_data += [img_arr]\n#                 cur_label = []\n#                 cur_label.append(train_df.loc[i, 'patient_overall'])\n#                 cur_label.append(train_df.loc[i,'C1'])\n#                 cur_label.append(train_df.loc[i,'C2'])\n#                 cur_label.append(train_df.loc[i,'C3'])\n#                 cur_label.append(train_df.loc[i,'C4'])\n#                 cur_label.append(train_df.loc[i,'C5'])\n#                 cur_label.append(train_df.loc[i,'C6'])\n#                 cur_label.append(train_df.loc[i,'C7'])\n                \n#                 train_label += [cur_label]\n#                 train_idx += [idx]\n                \n#                 if len(train_idx) == batch_size:\n#                     yield np.array(train_data), np.array(train_label)\n#                     train_data, train_label, train_idx = [], [], []\n                    \n#             i += 1\n            ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:15:46.031521Z","iopub.execute_input":"2022-08-23T20:15:46.031966Z","iopub.status.idle":"2022-08-23T20:15:46.044628Z","shell.execute_reply.started":"2022-08-23T20:15:46.03193Z","shell.execute_reply":"2022-08-23T20:15:46.04348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def test_generator(test_df, infinite = True, base_path = test_dir):\n#     while True:\n#         test_data = []\n#         test_idx = []\n        \n#         for i in range(len(test_df)):\n            \n#             if type(test_df) is list: \n#                 idx = test_df[i]\n#             else:\n#                 idx = test_df.loc[i, 'StudyInstanceUID']\n                \n#             path = os.path.join(test_dir, idx)\n            \n#             if os.path.exists(path):\n#                 for filename in os.listdir(path):\n#                     img_path = os.path.join(path, filename)\n                    \n#                     if not \"dcm\" in img_path:\n#                         continue\n                        \n#                     dc  = dicom.read_file(img_path)\n#                     if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#                         continue\n#                     img = load_dicom(img_path)\n#                     img = cv2.resize(img, (64, 64))\n#                     img_arr = img_to_array(img)\n#                     img_arr /= 255.0\n\n#                     test_data += [img_arr]\n#                     test_idx += [idx]\n\n#                     if len(test_idx) == batch_size:\n#                         yield np.array(test_data)\n#                         train_data = []\n#         print(len(test_data))2\n\n#         if not infinite:\n#             return np.array(test_data)\n#             break\n\n#         if len(test_data) > 0: \n#             yield np.array(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:15:46.04652Z","iopub.execute_input":"2022-08-23T20:15:46.047727Z","iopub.status.idle":"2022-08-23T20:15:46.058303Z","shell.execute_reply.started":"2022-08-23T20:15:46.047378Z","shell.execute_reply":"2022-08-23T20:15:46.057295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import BatchNormalization, Conv2D, MaxPooling2D, Dropout, Flatten, Dense, Activation","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.43875Z","iopub.execute_input":"2022-08-23T21:11:02.44059Z","iopub.status.idle":"2022-08-23T21:11:02.448487Z","shell.execute_reply.started":"2022-08-23T21:11:02.44056Z","shell.execute_reply":"2022-08-23T21:11:02.447249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_spape = (64, 64, 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.450267Z","iopub.execute_input":"2022-08-23T21:11:02.451065Z","iopub.status.idle":"2022-08-23T21:11:02.457851Z","shell.execute_reply.started":"2022-08-23T21:11:02.451028Z","shell.execute_reply":"2022-08-23T21:11:02.456823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install efficientnet\n# import efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:13:09.662445Z","iopub.execute_input":"2022-08-23T21:13:09.663162Z","iopub.status.idle":"2022-08-23T21:15:39.6603Z","shell.execute_reply.started":"2022-08-23T21:13:09.663122Z","shell.execute_reply":"2022-08-23T21:15:39.658746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_model():\n#     tf.keras.applications.efficientnet.EfficientNetB7(input_shape=input_spape,weights='imagenet',include_top=False),\n#     tf.keras.layers.GlobalAveragePooling2D(),\n#     tf.keras.layers.Dropout(0.25),\n#     tf.keras.layers.Dense(256, activation='swish'),\n#     tf.keras.layers.Dropout(0.25),\n#     tf.keras.layers.Dense(7, activation='sigmoid')\n    \n    \n#     model.compile(loss=\"binary_crossentropy\", \n#                   optimizer = tf.keras.optimizers.Nadam(learning_rate = 0.001),\n#                   metrics=[tf.keras.metrics.BinaryAccuracy(), tf.keras.metrics.AUC(multi_label=True)])\n#     model.summary()\n    \n#     return model","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.462603Z","iopub.execute_input":"2022-08-23T21:11:02.463618Z","iopub.status.idle":"2022-08-23T21:11:02.472003Z","shell.execute_reply.started":"2022-08-23T21:11:02.463591Z","shell.execute_reply":"2022-08-23T21:11:02.470941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(32, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.2))\n\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(64, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Dropout(0.2))\n\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(128, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25))\n    \n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(256, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25))\n    \n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(256, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25)) \n    \n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(128, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.25))   \n\n    model = tf.keras.models.Sequential()\n    model.add(BatchNormalization(input_shape = input_spape))\n    model.add(Conv2D(64, (5, 5), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n    model.add(Dropout(0.2))\n\n    model.add(Flatten())\n    model.add(Dense(1024))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(128))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(32))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(7))\n    model.add(Activation('softmax'))\n    \n    model.summary()\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Nadam(learning_rate = 0.001),\n                  metrics=[tf.keras.metrics.BinaryAccuracy(), tf.keras.metrics.AUC(multi_label=True)])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:15:46.083876Z","iopub.execute_input":"2022-08-23T20:15:46.08422Z","iopub.status.idle":"2022-08-23T20:15:46.103423Z","shell.execute_reply.started":"2022-08-23T20:15:46.084184Z","shell.execute_reply":"2022-08-23T20:15:46.101861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self, gen_df, batch_size=64, input_size=(64, 64), n_channels=3, n_classes=7, shuffle=True):\n        'Initialization'\n        self.gen_df = gen_df\n        self.list_IDs = gen_df['ID']\n        self.batch_size = batch_size\n        self.input_size = input_size\n        self.shuffle = shuffle\n        \n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        \n        self.on_epoch_end()\n        \n    def __get_input(self, path):\n        img = load_dicom(path)\n        img = cv2.resize(img, (64, 64))\n        img_arr = img_to_array(img)\n        img_arr /= 255.0\n        \n        return img_arr\n    \n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        \n        # generate indexes for the current batch\n        ID_indexes = self.indexes[index * self.batch_size : (index+1) * self.batch_size]\n        X = []\n        y = []\n        #generate the list of patient IDs\n        gen_df_temp = gen_df.iloc[ID_indexes, :]\n        #generate the data based on gen_df_temp\n        \n        for i in range(len(gen_df_temp)):\n            path = gen_df_temp.iloc[i, 1]\n            dc = dicom.read_file(path)\n            if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n                continue\n                \n            X.append(self.__get_input(path))\n            y.append(gen_df_temp.iloc[i, 2:].values.tolist())\n        \n        return np.array(X), np.array(y)\n    \n    def __len__(self):\n        return int(np.floor(len(self.list_IDs) // self.batch_size))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.474957Z","iopub.execute_input":"2022-08-23T21:11:02.476977Z","iopub.status.idle":"2022-08-23T21:11:02.490751Z","shell.execute_reply.started":"2022-08-23T21:11:02.476941Z","shell.execute_reply":"2022-08-23T21:11:02.489665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = TrainDataGen(train_gen_df)\nval_generator = TrainDataGen(val_gen_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.492187Z","iopub.execute_input":"2022-08-23T21:11:02.493359Z","iopub.status.idle":"2022-08-23T21:11:02.519746Z","shell.execute_reply.started":"2022-08-23T21:11:02.493323Z","shell.execute_reply":"2022-08-23T21:11:02.518853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:02.521328Z","iopub.execute_input":"2022-08-23T21:11:02.522004Z","iopub.status.idle":"2022-08-23T21:11:02.528774Z","shell.execute_reply.started":"2022-08-23T21:11:02.521965Z","shell.execute_reply":"2022-08-23T21:11:02.527705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n# tf.debugging.set_log_device_placement(False)\nx_train = train_df.reset_index()\n# x_val = train_df.iloc[val_idx].reset_index()\n\nmodel = create_model()\n# steps_per_epoch = int(len(train_gen_df) / 64)\nhist = model.fit(train_generator,\n                    epochs = 100,\n                    verbose = 1,\n                    callbacks = [tf.keras.callbacks.EarlyStopping(monitor=\"loss\", min_delta=0.005, \n                         patience=15)],\n    #                             validation_steps = max((len(x_val) // 64), 1),\n                    steps_per_epoch = 25,\n#                     validation_data = val_generator,\n#                     val_pred = model.predict(val_generator),\n#                                       steps = max((len(test_df) // 64), 1)\n                )\n\n# try: # the best we can do at the moment..\n#     preds = model.predict_generator(test_generator(test_df, min(len(test_df), 64), infinite = False, base_path = test_dir), steps = max((len(test_df) // 64), 1))\n\n#     new_preds = []\n#     for pred_idx in range(len(preds)):\n#         new_preds.append(preds[pred_idx][prediction_type_mapping[pred_idx]])\n#     # submission['fractured'] += preds[:, prediction_type_mapping] / 5\n#     submission['fractured'] += np.array(new_preds) / 5\n\n# except: traceback.print_exc()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T21:11:48.960948Z","iopub.execute_input":"2022-08-23T21:11:48.962063Z","iopub.status.idle":"2022-08-23T21:12:13.110399Z","shell.execute_reply.started":"2022-08-23T21:11:48.962013Z","shell.execute_reply":"2022-08-23T21:12:13.108629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_input(path):\n#     img = load_dicom(path)\n#     img = cv2.resize(img, (64, 64))\n#     img_arr = img_to_array(img)\n#     img_arr /= 255.0\n\n#     return img_arr\n\n# def get_val_data(val_gen_df):\n#     X = []\n#     y = []\n#     #generate the list of patient IDs\n#     gen_df_temp = val_gen_df\n#     #generate the data based on gen_df_temp\n\n#     for i in range(len(gen_df_temp)):\n#         path = gen_df_temp.iloc[i, 1]\n#         dc = dicom.read_file(path)\n#         if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#             continue\n\n#         X.append(get_input(path))\n#         y.append(gen_df_temp.iloc[i, 2:].values.tolist())\n\n#     return np.array(X), np.array(y)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:27:48.078236Z","iopub.execute_input":"2022-08-23T20:27:48.079458Z","iopub.status.idle":"2022-08-23T20:27:48.087475Z","shell.execute_reply.started":"2022-08-23T20:27:48.079407Z","shell.execute_reply":"2022-08-23T20:27:48.086245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_gen_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:28:10.471589Z","iopub.execute_input":"2022-08-23T20:28:10.472384Z","iopub.status.idle":"2022-08-23T20:28:10.485634Z","shell.execute_reply.started":"2022-08-23T20:28:10.472337Z","shell.execute_reply":"2022-08-23T20:28:10.484561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_X, val_y = get_val_data(val_gen_df.iloc[:1000, :])","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:38:51.032518Z","iopub.execute_input":"2022-08-23T20:38:51.033657Z","iopub.status.idle":"2022-08-23T20:39:03.41697Z","shell.execute_reply.started":"2022-08-23T20:38:51.033605Z","shell.execute_reply":"2022-08-23T20:39:03.415912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_pred = hist.model.predict(val_X)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:39:07.278808Z","iopub.execute_input":"2022-08-23T20:39:07.279199Z","iopub.status.idle":"2022-08-23T20:39:07.486207Z","shell.execute_reply.started":"2022-08-23T20:39:07.279165Z","shell.execute_reply":"2022-08-23T20:39:07.485036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:33:13.709782Z","iopub.execute_input":"2022-08-23T20:33:13.710526Z","iopub.status.idle":"2022-08-23T20:33:13.715509Z","shell.execute_reply.started":"2022-08-23T20:33:13.710485Z","shell.execute_reply":"2022-08-23T20:33:13.71438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_pred[val_pred >= 0.22] = 1\n# val_pred[val_pred < 0.22] = 0","metadata":{"execution":{"iopub.status.busy":"2022-08-23T20:39:26.840512Z","iopub.execute_input":"2022-08-23T20:39:26.841111Z","iopub.status.idle":"2022-08-23T20:39:26.847003Z","shell.execute_reply.started":"2022-08-23T20:39:26.841072Z","shell.execute_reply":"2022-08-23T20:39:26.845757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for train_idx, val_idx in StratifiedKFold(5).split(train_df, train_df['patient_overall']):\n#     tf.keras.backend.clear_session()\n#     x_train = train_df.iloc[train_idx].reset_index()\n#     x_val = train_df.iloc[val_idx].reset_index()\n    \n#     model = create_model()\n    \n#     hist = model.fit(train_generator(x_train, min(len(x_train), 64), infinite=False, base_path=train_dir),\n#                                 epochs = 50,\n#                                 verbose = 1,\n# #                                 callbacks = [tf.keras.callbacks.EarlyStopping(monitor=\"val_loss\", min_delta=0.001, \n# #                                      patience=10)],\n#                                 validation_steps = max((len(x_val) // 64), 1),\n#                                 steps_per_epoch = max((len(x_train) // 64), 1),\n#                                 validation_data = train_generator(x_val, min(len(x_val), 64), infinite = False, base_path = train_dir))\n#     val_pred = model.predict(train_generator(x_val, min(len(test_df), 64), infinite = False, base_path = train_dir), \n#                                       steps = max((len(test_df) // 64), 1))\n    \n#     try: # the best we can do at the moment..\n#         preds = model.predict_generator(test_generator(test_df, min(len(test_df), 64), infinite = False, base_path = test_dir), steps = max((len(test_df) // 64), 1))\n        \n#         new_preds = []\n#         for pred_idx in range(len(preds)):\n#             new_preds.append(preds[pred_idx][prediction_type_mapping[pred_idx]])\n#         # submission['fractured'] += preds[:, prediction_type_mapping] / 5\n#         submission['fractured'] += np.array(new_preds) / 5\n        \n#     except: traceback.print_exc()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_dir='../input/rsna-2022-cervical-spine-fracture-detection/test_images'\n# testset=[]\n# testidt=[]\n# for i in tqdm(range(len(test_df))):\n#     idt=test_df[i]\n#     path=os.path.join(test_dir,idt)   \n    \n#     for im in os.listdir(path):\n#         dc = dicom.read_file(os.path.join(path,im))\n        \n#         if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#             continue\n#         img=load_dicom(os.path.join(path,im)) \n\n#         img=cv.resize(img,(64,64)) \n#         image=img_to_array(img)\n#         image=image/255.0\n#         testset+=[image]\n#         testidt+=[idt]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist.model.save('model_01')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n\n# plt.plot(hist.history['loss'])\n# # plt.plot(hist.history['binary_accuracy'])\n# plt.title('model accuracy')\n# plt.ylabel('binary_accuracy')\n# plt.xlabel('epoch')\n# plt.legend(['train', 'test'], loc='upper left')\n# plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! ls /kaggle/working","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a href=\"./model_01\"> Download File </a>","metadata":{}},{"cell_type":"code","source":"# test_data = []\n# test_idx = []\n\n# for i in range(len(test_df)):\n\n#     if type(test_df) is list: \n#         idx = test_df[i]\n#     else:\n#         idx = test_df.loc[i, 'StudyInstanceUID']\n\n#     path = os.path.join(test_dir, idx)\n#     print(path)\n#     if os.path.exists(path):\n#         for filename in os.listdir(path):\n#             img_path = os.path.join(path, filename)\n\n#             if not \"dcm\" in img_path:\n#                 continue\n\n#             dc  = dicom.read_file(img_path)\n#             if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n#                 continue\n#             img = load_dicom(img_path)\n#             img = cv2.resize(img, (64, 64))\n#             img_arr = img_to_array(img)\n#             img_arr /= 255.0\n\n#             test_data += [img_arr]\n#             test_idx += [idx]\n            \n#     else:\n#         print(\"problem with the path\")\n            \n# print(len(test_data))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds = hist.model.predict(test_generator(test_df, min(len(test_df), 64), infinite = False, base_path = test_dir), steps = max((len(test_df) // 64)))\n\n# new_preds = []\n# for pred_idx in range(len(preds)):\n#     new_preds.append(preds[pred_idx][prediction_type_mapping[pred_idx]])\n# # submission['fractured'] += preds[:, prediction_type_mapping] / 5\n# submission['fractured'] += np.array(new_preds) / 5","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tf.keras.backend.clear_session()\n\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# # instantiate a distribution strategy\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\n# print(\"All devices: \", tf.config.list_logical_devices('TPU'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tf.keras.utils.plot_model(model, show_shapes=True, rankdir='TB')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = create_model()\n# model.compile(\n#   optimizer=tf.keras.optimizers.Nadam(learning_rate=0.0005),\n#   loss='binary_crossentropy',\n#   metrics=['accuracy'])\n\n# callbacks = [\n#     # TensorBoard will store logs for each epoch and graph performance for us.\n#     # tf.keras.callbacks.TensorBoard(log_dir=log_dir, histogram_freq=1),\n#     # ModelCheckpoint will save models after each epoch for retrieval later.\n#     # tf.keras.callbacks.ModelCheckpoint(checkpoint_path),\n#     # EarlyStopping will terminate training when val_loss ceases to improve.\n#     tf.keras.callbacks.EarlyStopping(monitor=\"loss\", min_delta=0.001, \n#                                      patience=10, restore_best_weights=True)\n# ]\n\n\n# model.fit(\n#     X_train, Y_train,\n#     epochs=100,\n#     batch_size=64,\n#     callbacks=callbacks\n# )\n\n# model.save_weights('./fashion_mnist.h5', overwrite=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred=model.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result = pd.DataFrame(columns = train_df.columns, index = range(len(testidt)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in tqdm(range(len(testidt))):\n#     result.loc[i, 'StudyInstanceUID'] = testidt[i]\n#     rows = np.int64(y_pred[i]>0.07)\n#     result.loc[i, 'patient_overall'] = int(bool(np.sum(rows)))\n#     result.loc[i, 'C1'] = rows[0]\n#     result.loc[i, 'C2'] = rows[1]\n#     result.loc[i, 'C3'] = rows[2]\n#     result.loc[i, 'C4'] = rows[3]\n#     result.loc[i, 'C5'] = rows[4]\n#     result.loc[i, 'C6'] = rows[5]\n#     result.loc[i, 'C7'] = rows[6]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# means = result[['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']].mean().to_dict()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df2 = pd.read_csv('../input/rsna-2022-cervical-spine-fracture-detection/test.csv')\n# # means = train_df.mean(numeric_only=True).to_dict()\n# test_df2['fractured'] = test_df2['prediction_type'].map(means)\n\n# test_df2[['row_id','fractured']].to_csv('submission.csv', index=False, float_format='%.1g')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}