{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport pydicom\nimport matplotlib.pyplot as plt\nimport scipy.io\nimport numpy as np\nimport cv2\nfrom PIL import Image\nimport pandas as pd\nimport gc\nfrom tqdm import tqdm\n\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import applications\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.layers import Dense, Conv2D , MaxPool2D , Flatten , Dropout , BatchNormalization\n\nfrom sklearn.model_selection import RepeatedKFold, cross_val_score, train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-03-30T20:32:33.903822Z","iopub.execute_input":"2022-03-30T20:32:33.904577Z","iopub.status.idle":"2022-03-30T20:32:40.366642Z","shell.execute_reply.started":"2022-03-30T20:32:33.90444Z","shell.execute_reply":"2022-03-30T20:32:40.365846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/0003b3d648eb/d2b2960c2bbf/00ac73cfc372.dcm\")\ndcm_sample=ds.pixel_array.astype('float32')\nscaled_image = (np.maximum(dcm_sample, 0) / dcm_sample.max())\nplt.imshow(scaled_image)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T20:32:40.369884Z","iopub.execute_input":"2022-03-30T20:32:40.370136Z","iopub.status.idle":"2022-03-30T20:32:40.752255Z","shell.execute_reply.started":"2022-03-30T20:32:40.370102Z","shell.execute_reply":"2022-03-30T20:32:40.751492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#not_noraml\ndf = pd.read_csv(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train.csv\")\ndf = df.loc[df[\"pe_present_on_image\"]==1,:].reset_index(drop=True)\nprint(len(df))\ndf.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T20:32:40.75339Z","iopub.execute_input":"2022-03-30T20:32:40.753665Z","iopub.status.idle":"2022-03-30T20:32:44.669095Z","shell.execute_reply.started":"2022-03-30T20:32:40.753628Z","shell.execute_reply":"2022-03-30T20:32:44.668392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#normal\ndf1 = pd.read_csv(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train.csv\")\ndf1 = df1.loc[(df1[\"pe_present_on_image\"] == 0) & (df1[\"negative_exam_for_pe\"] == 1) ,:].reset_index(drop=True)\nprint(len(df1))\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T20:32:44.672094Z","iopub.execute_input":"2022-03-30T20:32:44.672297Z","iopub.status.idle":"2022-03-30T20:32:47.406028Z","shell.execute_reply.started":"2022-03-30T20:32:44.67227Z","shell.execute_reply":"2022-03-30T20:32:47.405332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/data\n\n!mkdir /kaggle/data/train\n!mkdir /kaggle/data/valid\n!mkdir /kaggle/data/test\n\n!mkdir /kaggle/data/train/normal\n!mkdir /kaggle/data/train/not_normal\n\n!mkdir /kaggle/data/valid/normal\n!mkdir /kaggle/data/valid/not_normal\n\n!mkdir /kaggle/data/test/normal\n!mkdir /kaggle/data/test/not_normal","metadata":{"execution":{"iopub.status.busy":"2022-03-30T20:32:47.407136Z","iopub.execute_input":"2022-03-30T20:32:47.407812Z","iopub.status.idle":"2022-03-30T20:32:54.273701Z","shell.execute_reply.started":"2022-03-30T20:32:47.407774Z","shell.execute_reply":"2022-03-30T20:32:54.27273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train not_normal\nfor i in tqdm(range(10000)):\n    dcm = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/\"+df.loc[i,'StudyInstanceUID']+'/'+df.loc[i,'SeriesInstanceUID']+'/'+df.loc[i,'SOPInstanceUID']+'.dcm')\n    dc_image=dcm.pixel_array.astype('float32')\n    #scaled_image = (np.maximum(dc_image, 0) / dc_image.max())\n    #scaled_image = np.reshape(scaled_image,(scaled_image.shape[0], scaled_image.shape[1], 1))\n    im = Image.fromarray(dc_image).convert('RGB').resize((256,256))  \n    im.save(\"/kaggle/data/train/not_normal/\"+str(i)+\".jpg\")\n    del dcm, dc_image, im\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T20:32:54.275357Z","iopub.execute_input":"2022-03-30T20:32:54.275783Z","iopub.status.idle":"2022-03-30T21:04:48.580814Z","shell.execute_reply.started":"2022-03-30T20:32:54.275737Z","shell.execute_reply":"2022-03-30T21:04:48.578882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train normal\nfor i in tqdm(range(10000)):\n    dcm = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/\"+df1.loc[i,'StudyInstanceUID']+'/'+df1.loc[i,'SeriesInstanceUID']+'/'+df1.loc[i,'SOPInstanceUID']+'.dcm')\n    dc_image = dcm.pixel_array.astype('float32')\n    #scaled_image = (np.maximum(dc_image, 0) / dc_image.max())\n    #scaled_image = np.reshape(scaled_image,(scaled_image.shape[0], scaled_image.shape[1], 1))\n    im = Image.fromarray(dc_image).convert('RGB').resize((256,256))  \n    im.save(\"/kaggle/data/train/normal/\"+str(i)+\".jpg\")\n    del dcm, dc_image, im\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T21:04:48.58199Z","iopub.execute_input":"2022-03-30T21:04:48.582573Z","iopub.status.idle":"2022-03-30T21:36:55.516484Z","shell.execute_reply.started":"2022-03-30T21:04:48.582535Z","shell.execute_reply":"2022-03-30T21:36:55.515211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Valid not_normal\nfor i in tqdm(range(10000,12000)):\n    dcm = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/\"+df.loc[i,'StudyInstanceUID']+'/'+df.loc[i,'SeriesInstanceUID']+'/'+df.loc[i,'SOPInstanceUID']+'.dcm')\n    dc_image=dcm.pixel_array.astype('float32')\n    #scaled_image = (np.maximum(dc_image, 0) / dc_image.max())\n    #scaled_image = np.reshape(scaled_image,(scaled_image.shape[0], scaled_image.shape[1], 1))\n    im = Image.fromarray(dc_image).convert('RGB').resize((256,256))  \n    im.save(\"/kaggle/data/valid/not_normal/\"+str(i)+\".jpg\")\n    del dcm, dc_image, im\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T21:36:55.518594Z","iopub.execute_input":"2022-03-30T21:36:55.519365Z","iopub.status.idle":"2022-03-30T21:43:23.660473Z","shell.execute_reply.started":"2022-03-30T21:36:55.51932Z","shell.execute_reply":"2022-03-30T21:43:23.659783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Valid normal\nfor i in tqdm(range(10000,12000)):\n    dcm = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/\"+df1.loc[i,'StudyInstanceUID']+'/'+df1.loc[i,'SeriesInstanceUID']+'/'+df1.loc[i,'SOPInstanceUID']+'.dcm')\n    dc_image=dcm.pixel_array.astype('float32')\n    #scaled_image = (np.maximum(dc_image, 0) / dc_image.max())\n    #scaled_image = np.reshape(scaled_image,(scaled_image.shape[0], scaled_image.shape[1], 1))\n    im = Image.fromarray(dc_image).convert('RGB').resize((256,256))  \n    im.save(\"/kaggle/data/valid/normal/\"+str(i)+\".jpg\")\n    del dcm, dc_image, im\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T21:43:23.662649Z","iopub.execute_input":"2022-03-30T21:43:23.663869Z","iopub.status.idle":"2022-03-30T21:49:49.22711Z","shell.execute_reply.started":"2022-03-30T21:43:23.663828Z","shell.execute_reply":"2022-03-30T21:49:49.226278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test not_normal\nfor i in tqdm(range(12000,14000)):\n    dcm = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/\"+df.loc[i,'StudyInstanceUID']+'/'+df.loc[i,'SeriesInstanceUID']+'/'+df.loc[i,'SOPInstanceUID']+'.dcm')\n    dc_image=dcm.pixel_array.astype('float32')\n    #scaled_image = (np.maximum(dc_image, 0) / dc_image.max())\n    #scaled_image = np.reshape(scaled_image,(scaled_image.shape[0], scaled_image.shape[1], 1))\n    im = Image.fromarray(dc_image).convert('RGB').resize((256,256))  \n    im.save(\"/kaggle/data/test/not_normal/\"+str(i)+\".jpg\")\n    del dcm, dc_image, im\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T21:49:49.232415Z","iopub.execute_input":"2022-03-30T21:49:49.232709Z","iopub.status.idle":"2022-03-30T21:56:15.333066Z","shell.execute_reply.started":"2022-03-30T21:49:49.232672Z","shell.execute_reply":"2022-03-30T21:56:15.332387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test normal\nfor i in tqdm(range(12000,14000)):\n    dcm = pydicom.dcmread(\"/kaggle/input/rsna-str-pulmonary-embolism-detection/train/\"+df1.loc[i,'StudyInstanceUID']+'/'+df1.loc[i,'SeriesInstanceUID']+'/'+df1.loc[i,'SOPInstanceUID']+'.dcm')\n    dc_image=dcm.pixel_array.astype('float32')\n    #scaled_image = (np.maximum(dc_image, 0) / dc_image.max())\n    #scaled_image = np.reshape(scaled_image,(scaled_image.shape[0], scaled_image.shape[1], 1))\n    im = Image.fromarray(dc_image).convert('RGB').resize((256,256))  \n    im.save(\"/kaggle/data/test/normal/\"+str(i)+\".jpg\")\n    del dcm, dc_image, im\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T21:56:15.424999Z","iopub.execute_input":"2022-03-30T21:56:15.425412Z","iopub.status.idle":"2022-03-30T22:02:40.011199Z","shell.execute_reply.started":"2022-03-30T21:56:15.425364Z","shell.execute_reply":"2022-03-30T22:02:40.010365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n      featurewise_center=False,  \n      samplewise_center=False, \n      featurewise_std_normalization=False,  \n      samplewise_std_normalization=False, \n      rescale=1./255,\n      rotation_range=20,\n      width_shift_range=0.2,\n      height_shift_range=0.2,\n      shear_range=0.2,\n      zoom_range=0.2,\n      horizontal_flip=True,\n      vertical_flip=True,\n      fill_mode='nearest')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:40.013125Z","iopub.execute_input":"2022-03-30T22:02:40.013463Z","iopub.status.idle":"2022-03-30T22:02:40.02186Z","shell.execute_reply.started":"2022-03-30T22:02:40.013425Z","shell.execute_reply":"2022-03-30T22:02:40.020425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_directory(\n        '/kaggle/data/train',\n        target_size=(256, 256),\n        batch_size=64,\n        class_mode='binary')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:40.023933Z","iopub.execute_input":"2022-03-30T22:02:40.024879Z","iopub.status.idle":"2022-03-30T22:02:40.614226Z","shell.execute_reply.started":"2022-03-30T22:02:40.024841Z","shell.execute_reply":"2022-03-30T22:02:40.613461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_datagen = ImageDataGenerator(\n      rescale=1./255)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:40.616473Z","iopub.execute_input":"2022-03-30T22:02:40.617622Z","iopub.status.idle":"2022-03-30T22:02:40.622844Z","shell.execute_reply.started":"2022-03-30T22:02:40.617582Z","shell.execute_reply":"2022-03-30T22:02:40.622117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_generator = valid_datagen.flow_from_directory(\n        '/kaggle/data/valid',\n        target_size=(256, 256),\n        batch_size=64,\n        class_mode='binary')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:40.625102Z","iopub.execute_input":"2022-03-30T22:02:40.626373Z","iopub.status.idle":"2022-03-30T22:02:40.845485Z","shell.execute_reply.started":"2022-03-30T22:02:40.626333Z","shell.execute_reply":"2022-03-30T22:02:40.844573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(\n      rescale=1./255)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:40.847788Z","iopub.execute_input":"2022-03-30T22:02:40.848063Z","iopub.status.idle":"2022-03-30T22:02:40.853702Z","shell.execute_reply.started":"2022-03-30T22:02:40.848024Z","shell.execute_reply":"2022-03-30T22:02:40.852918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = valid_datagen.flow_from_directory(\n        '/kaggle/data/test',\n        target_size=(256, 256),\n        batch_size=64,\n        class_mode='binary')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:40.85597Z","iopub.execute_input":"2022-03-30T22:02:40.857311Z","iopub.status.idle":"2022-03-30T22:02:41.07825Z","shell.execute_reply.started":"2022-03-30T22:02:40.85727Z","shell.execute_reply":"2022-03-30T22:02:41.077374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\ntpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:02:41.079491Z","iopub.execute_input":"2022-03-30T22:02:41.08168Z","iopub.status.idle":"2022-03-30T22:02:41.08953Z","shell.execute_reply.started":"2022-03-30T22:02:41.081636Z","shell.execute_reply":"2022-03-30T22:02:41.088729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32 , (3,3) , strides = 1 , padding = 'same' , activation = 'relu' , input_shape = ( 256, 256, 3)))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(64 , (3,3) , strides = 1 , padding = 'same' , activation = 'relu'))\nmodel.add(Dropout(0.1))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(64 , (3,3) , strides = 1 , padding = 'same' , activation = 'relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(128 , (3,3) , strides = 1 , padding = 'same' , activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Conv2D(256 , (3,3) , strides = 1 , padding = 'same' , activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D((2,2) , strides = 2 , padding = 'same'))\nmodel.add(Flatten())\nmodel.add(Dense(units = 128 , activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units = 1 , activation = 'sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:39:20.117628Z","iopub.execute_input":"2022-03-30T22:39:20.117888Z","iopub.status.idle":"2022-03-30T22:39:20.285128Z","shell.execute_reply.started":"2022-03-30T22:39:20.117859Z","shell.execute_reply":"2022-03-30T22:39:20.284352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/models","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:29:11.279619Z","iopub.execute_input":"2022-03-30T22:29:11.280263Z","iopub.status.idle":"2022-03-30T22:29:12.041045Z","shell.execute_reply.started":"2022-03-30T22:29:11.280225Z","shell.execute_reply":"2022-03-30T22:29:12.039997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = tf.keras.losses.BinaryCrossentropy()\nmodel.compile(loss=loss, \n              optimizer='Adam', \n              metrics=['binary_accuracy'])\n\nlearning_rate_reduction = ReduceLROnPlateau(monitor='val_binary_accuracy', patience = 2, verbose=1,factor=0.3, min_lr=0.000001)\n\nfilepath = \"/kaggle/models/saved-model-{epoch:02d}-{val_binary_accuracy:.2f}.hdf5\"\ncheckpoint = ModelCheckpoint(filepath, monitor='val_loss', verbose=1, \n                             save_best_only=False,save_freq='epoch')\n","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:39:22.778416Z","iopub.execute_input":"2022-03-30T22:39:22.779012Z","iopub.status.idle":"2022-03-30T22:39:22.794935Z","shell.execute_reply.started":"2022-03-30T22:39:22.778974Z","shell.execute_reply":"2022-03-30T22:39:22.793998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n      train_generator,\n      epochs=25,\n      validation_data=valid_generator,\n      validation_steps=4,\n      callbacks=[checkpoint,learning_rate_reduction],\n      verbose=1,\n    \n      )","metadata":{"execution":{"iopub.status.busy":"2022-03-30T22:39:24.352003Z","iopub.execute_input":"2022-03-30T22:39:24.352268Z","iopub.status.idle":"2022-03-31T00:23:35.500057Z","shell.execute_reply.started":"2022-03-30T22:39:24.352241Z","shell.execute_reply":"2022-03-31T00:23:35.495335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['binary_accuracy'], label='The score of correct predictions on the training set')\nplt.plot(history.history['val_binary_accuracy'], label='The score of correct predictions on the val set')\nplt.xlabel('Epoch')\nplt.ylabel('Score correct answers')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-31T00:23:35.502988Z","iopub.execute_input":"2022-03-31T00:23:35.503609Z","iopub.status.idle":"2022-03-31T00:23:35.756956Z","shell.execute_reply.started":"2022-03-31T00:23:35.50357Z","shell.execute_reply":"2022-03-31T00:23:35.754581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_acc = 0\nbest_model = \"\"\nfor i in os.listdir(\"/kaggle/models\"):\n    model.load_weights(\"/kaggle/models/\"+i)\n    loss, acc = model.evaluate_generator(test_generator, steps=3, verbose=0)\n    if acc > best_acc:\n        best_model = i\n        best_acc = acc","metadata":{"execution":{"iopub.status.busy":"2022-03-31T00:23:56.898496Z","iopub.execute_input":"2022-03-31T00:23:56.899678Z","iopub.status.idle":"2022-03-31T00:24:16.978947Z","shell.execute_reply.started":"2022-03-31T00:23:56.899636Z","shell.execute_reply":"2022-03-31T00:24:16.978143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights(\"/kaggle/models/\"+best_model)\nloss, acc = model.evaluate_generator(test_generator, steps=3, verbose=0)\nacc = acc *100\nprint(f\"accuracy is: {acc:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2022-03-31T00:27:27.294627Z","iopub.execute_input":"2022-03-31T00:27:27.29547Z","iopub.status.idle":"2022-03-31T00:27:28.263087Z","shell.execute_reply.started":"2022-03-31T00:27:27.295434Z","shell.execute_reply":"2022-03-31T00:27:28.262217Z"},"trusted":true},"execution_count":null,"outputs":[]}]}