{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pydicom\nimport seaborn as sns\nimport imageio\nfrom IPython import display\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Add Reference here"},{"metadata":{},"cell_type":"markdown","source":"* https://www.kaggle.com/nitindatta/pulmonary-embolism-dicom-preprocessing-eda\n* https://www.kaggle.com/omarkhald/pe-detection"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/rsna-str-pulmonary-embolism-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/rsna-str-pulmonary-embolism-detection/test.csv')\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.shape, df_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head().T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of unique Study instances are\", df_train['StudyInstanceUID'].nunique())\nprint(\"Number of unique Series instances are\", df_train['SeriesInstanceUID'].nunique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Null values in train data:',df_train.isnull().sum().sum())\nprint('Null values in test data:',df_test.isnull().sum().sum())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Display the single DICOM image"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pydicom\nfrom pydicom.data import get_testdata_files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset = pydicom.dcmread('/kaggle/input/rsna-str-pulmonary-embolism-detection/train/b4548bee81e8/ac1aea5d7662/cc96a7a2e72c.dcm')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset.pixel_array","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset.PixelData[0:40]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Normal mode:\nprint(\"Modality.........:\", dataset.Modality)\n\nif 'PixelData' in dataset:\n    rows = int(dataset.Rows)\n    cols = int(dataset.Columns)\n    print(\"Image size.......: {rows:d} x {cols:d}, {size:d} bytes\".format(\n        rows=rows, cols=cols, size=len(dataset.PixelData)))\n    if 'PixelSpacing' in dataset:\n        print(\"Pixel spacing....:\", dataset.PixelSpacing)\n\n# plot the image using matplotlib\nplt.imshow(dataset.pixel_array, cmap=plt.cm.bone)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> ### Display multiple DICOM images"},{"metadata":{"trusted":true},"cell_type":"code","source":"from os import listdir, mkdir","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"basepath = \"../input/rsna-str-pulmonary-embolism-detection/\"\nlistdir(basepath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndcmfile_path = basepath + \"train/\" + df_train.StudyInstanceUID.values[0] +'/'+ df_train.SeriesInstanceUID.values[0]\nfiles = listdir(dcmfile_path)\nscans = [pydicom.dcmread(dcmfile_path + \"/\" + str(file)) for file in files]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(scans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=plt.figure(figsize=(8, 8))\ncolumns = 4\nrows = 4\nfor i in range(1, columns*rows +1):\n    img = scans[i].pixel_array \n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Understanding more about Hounsfield Units"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(12,6))\nfor n in range(10):\n    image = scans[n].pixel_array.flatten()\n    sns.distplot(image);\nplt.title(\"HU unit distributions for 10 examples\");","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 3D plotting the scan"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_slice(path):\n    slices = [pydicom.read_file(path + '/' + s) for s in listdir(path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices\n\ndef transform_to_hu(slices):\n    images = np.stack([file.pixel_array for file in slices])\n    images = images.astype(np.int16)\n\n    # convert ouside pixel-values to air:\n    # I'm using <= -1000 to be sure that other defaults are captured as well\n    images[images <= -1000] = 0\n    \n    # convert to HU\n    for n in range(len(slices)):\n        \n        intercept = slices[n].RescaleIntercept\n        slope = slices[n].RescaleSlope\n        \n        if slope != 1:\n            images[n] = slope * images[n].astype(np.float64)\n            images[n] = images[n].astype(np.int16)\n            \n        images[n] += np.int16(intercept)\n    \n    return np.array(images, dtype=np.int16)\n\ndef resample(image, scan, new_spacing=[1,1,1]):\n    spacing = np.array([float(scans_0[0].SliceThickness), \n                        float(scans_0[0].PixelSpacing[0]), \n                        float(scans_0[0].PixelSpacing[0])])\n\n\n    resize_factor = spacing / new_spacing\n    new_real_shape = image.shape * resize_factor\n    new_shape = np.round(new_real_shape)\n    real_resize_factor = new_shape / image.shape\n    new_spacing = spacing / real_resize_factor\n    \n    image = scipy.ndimage.interpolation.zoom(image, real_resize_factor)\n    \n    return image, new_spacing\n\ndef make_mesh(image, threshold=-300, step_size=1):\n    p = image.transpose(2,1,0)\n    verts, faces, norm, val = measure.marching_cubes_lewiner(p, threshold, step_size=step_size, allow_degenerate=True)\n    return verts, faces\n\n\ndef plt_3d(verts, faces):\n    print(\"Drawing\")\n    x,y,z = zip(*verts) \n    fig = plt.figure(figsize=(10, 10))\n    ax = fig.add_subplot(111, projection='3d')\n\n    # Fancy indexing: `verts[faces]` to generate a collection of triangles\n    mesh = Poly3DCollection(verts[faces], linewidths=0.05, alpha=1)\n    face_color = [1, 1, 0.9]\n    mesh.set_facecolor(face_color)\n    ax.add_collection3d(mesh)\n\n    ax.set_xlim(0, max(x))\n    ax.set_ylim(0, max(y))\n    ax.set_zlim(0, max(z))\n#     ax.set_axis_bgcolor((0.7, 0.7, 0.7))\n    ax.set_facecolor((0.7,0.7,0.7))\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"first_patient = load_slice('../input/rsna-str-pulmonary-embolism-detection/train/0003b3d648eb/d2b2960c2bbf')\nfirst_patient_pixels = transform_to_hu(first_patient)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imageio.mimsave(\"/tmp/gif.gif\", first_patient_pixels, duration=0.1)\ndisplay.Image(filename=\"/tmp/gif.gif\", format='png')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['filename'] = df_train[['StudyInstanceUID', 'SeriesInstanceUID', 'SOPInstanceUID']].apply(\n    lambda x: '/'.join(x.astype(str)),\n    axis=1\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['filename'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.utils import Sequence\nfrom skimage.transform import resize\nimport math\nclass generator(Sequence):\n    \n    def __init__(self,df,images_path,batch_size=32, image_size=256, shuffle=True):\n        self.df=df\n        self.images_path = images_path\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.shuffle = shuffle\n        self.nb_iteration = math.ceil((self.df.shape[0])/self.batch_size)\n        self.on_epoch_end()\n        \n    def load_img(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(filename).pixel_array\n        img= resize(img,(self.image_size, self.image_size))\n        img = img.reshape((self.image_size, self.image_size, 1))\n        np.stack([img, img, img], axis=2).reshape((self.image_size, self.image_size, 3))\n        return img\n        \n    def __getitem__(self, index):\n        # select batch\n        indicies = list(range(index*self.batch_size, min((index*self.batch_size)+self.batch_size ,(self.df.shape[0]))))\n        \n        images = []\n        for img_path in self.df['filename'].iloc[indicies].tolist():\n            img_path = img_path+\".dcm\"\n            img = self.load_img(os.path.join(self.images_path,img_path))\n            images.append(img)\n        y = self.df[['negative_exam_for_pe', 'rv_lv_ratio_gte_1', 'rv_lv_ratio_lt_1',\n                     'leftsided_pe', 'chronic_pe', 'rightsided_pe',\n                     'acute_and_chronic_pe', 'central_pe', 'indeterminate']].iloc[indicies].values\n        return np.array(images), np.array(y)\n         \n    def on_epoch_end(self):\n        if self.shuffle:\n            self.df=self.df.sample(frac=1)\n        \n    def __len__(self):\n        return self.nb_iteration","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_path=\"../input/rsna-str-pulmonary-embolism-detection/train/\"\ndf_train= df_train.iloc[:20000]\ndf_val= df_train.iloc[20000:25000]\ntrain_dataloader =  generator(df_train,images_path)\nval_dataloader =  generator(df_val,images_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x,y = next(enumerate(train_dataloader))[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\ninputs = Input((256, 256, 1))\nDensenet_model = tf.keras.applications.DenseNet121(\n            include_top=False,\n            weights=None,\n            input_shape=(256,256,1))\n\noutputs = Densenet_model(inputs)\noutputs = GlobalAveragePooling2D()(outputs)\noutputs = Dropout(0.25)(outputs)\noutputs = Dense(1024, activation='relu')(outputs)\noutputs = Dropout(0.25)(outputs)\noutputs = Dense(256, activation='relu')(outputs)\noutputs = Dropout(0.25)(outputs)\noutputs = Dense(64, activation='relu')(outputs)\nnepe = Dense(1, activation='sigmoid', name='negative_exam_for_pe')(outputs)\nrlrg1 = Dense(1, activation='sigmoid', name='rv_lv_ratio_gte_1')(outputs)\nrlrl1 = Dense(1, activation='sigmoid', name='rv_lv_ratio_lt_1')(outputs) \nlspe = Dense(1, activation='sigmoid', name='leftsided_pe')(outputs)\ncpe = Dense(1, activation='sigmoid', name='chronic_pe')(outputs)\nrspe = Dense(1, activation='sigmoid', name='rightsided_pe')(outputs)\naacpe = Dense(1, activation='sigmoid', name='acute_and_chronic_pe')(outputs)\ncnpe = Dense(1, activation='sigmoid', name='central_pe')(outputs)\nindt = Dense(1, activation='sigmoid', name='indeterminate')(outputs)\n\nmodel = Model(inputs=inputs, outputs={'negative_exam_for_pe':nepe,\n                                      'rv_lv_ratio_gte_1':rlrg1,\n                                      'rv_lv_ratio_lt_1':rlrl1,\n                                      'leftsided_pe':lspe,\n                                      'chronic_pe':cpe,\n                                      'rightsided_pe':rspe,\n                                      'acute_and_chronic_pe':aacpe,\n                                      'central_pe':cnpe,\n                                      'indeterminate':indt})\n\n\nmodel.compile(optimizer=Adam(lr=1e-3),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from tensorflow.keras.utils import plot_model\n#plot_model(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#hist = model.fit_generator( train_dataloader,validation_data = val_dataloader,epochs = 5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}