{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"# Codes from this cell are adopted from Quadcore/Richard Epstein public notebook\n# This notebook loads GDCM without Internet access.\n# GDCM is needed to read some DICOM compressed images.\n# Once you run a notebook and get the GDCM error, you must restart that Kernel to read the files, even if you load the GDCM software.\n# Note that you do not \"import GDCM\". You just \"import pydicom\".\n# The Dataset (gdcm-conda-install) was provided by Ronaldo S.A. Batista. Definitely deserves an upvote!\n\n!cp ../input/gdcm-conda-install/gdcm.tar .\n!tar -xvzf gdcm.tar\n!conda install --offline ./gdcm/gdcm-2.8.9-py37h71b2a6d_0.tar.bz2\n\nprint(\"GDCM installed.\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np, pandas as pd, os\nimport matplotlib.pyplot as plt\nimport glob\nimport datetime\nimport torch\nimport torchvision.transforms as transforms\nimport pydicom\nfrom pydicom import dcmread\nfrom tqdm import tqdm\nfrom typing import Dict\n\nimport cv2\nimport torch\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nimport albumentations as albu\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nstartTime = datetime.datetime.now()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def to_device(x, cuda_id=0):\n    return x.cuda(cuda_id) if torch.cuda.is_available() else x\n\n\ndef load_jit_model(x, cuda_id=0):\n    return torch.jit.load(x, map_location=f'cuda:{cuda_id}' if torch.cuda.is_available() else 'cpu')\n\n\nss = pd.read_csv('../input/rsna-str-pulmonary-embolism-detection/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class ClassificationDataset(Dataset):\n    def __init__(self,\n                 fold: int = 0,\n                 mode: str = 'test'):\n        self.mode = mode\n        self.df = pd.read_csv('../input/rsna-str-pulmonary-embolism-detection/test.csv')\n\n        self.labels = ['pe_present_on_image']\n\n    def __len__(self):\n        return len(self.df)\n\n    def transform(self, image):\n        normalize = albu.Normalize(\n              mean=[0.485, 0.456, 0.406],\n              std=[0.229, 0.224, 0.225]\n         )\n        pipeline = {\n            \n            \"test\": albu.Compose(\n                [\n                    albu.Resize(\n                        512,\n                        512,\n                    ),\n                ],\n                p=1.0,\n            ),\n        }\n\n        result = pipeline[self.mode](image=image)\n        return result[\"image\"]\n\n    @staticmethod\n    def _preprocess_img(img):\n        img = np.transpose(img, (2, 0, 1))\n        return img\n\n\n    @staticmethod\n    def window_image(img, window_center, window_width, intercept, slope, rescale=True):\n\n        img = (img * slope + intercept)\n        img_min = window_center - window_width // 2\n        img_max = window_center + window_width // 2\n        img[img < img_min] = img_min\n        img[img > img_max] = img_max\n\n        if rescale:\n            # Extra rescaling to 0-1, not in the original notebook\n            img = (img - img_min) / (img_max - img_min)\n\n        return img\n\n    @staticmethod\n    def get_first_of_dicom_field_as_int(x):\n        # get x[0] as in int is x is a 'pydicom.multival.MultiValue', otherwise get int(x)\n        if type(x) == pydicom.multival.MultiValue:\n            return int(x[0])\n        else:\n            return int(x)\n\n    def get_windowing(self, data):\n        dicom_fields = [data[('0028', '1050')].value,  # window center\n                        data[('0028', '1051')].value,  # window width\n                        data[('0028', '1052')].value,  # intercept\n                        data[('0028', '1053')].value]  # slope\n        return [self.get_first_of_dicom_field_as_int(x) for x in dicom_fields]\n\n    def __getitem__(self, item):\n        data = self.df.iloc[item]\n        try:\n            dcm = pydicom.dcmread(os.path.join('../input/rsna-str-pulmonary-embolism-detection/test/',\n                                               data.StudyInstanceUID,\n                                               data.SeriesInstanceUID,\n                                               f'{data.SOPInstanceUID}.dcm'))\n            window_center, window_width, intercept, slope = self.get_windowing(dcm)\n            img = dcm.pixel_array\n            image1 = np.expand_dims(self.window_image(img, -600, 1500, intercept, slope), axis=-1)  # LUNG window\n            image2 = np.expand_dims(self.window_image(img, 100, 700, intercept, slope), axis=-1)  # PE window\n            image3 = np.expand_dims(self.window_image(img, 40, 400, intercept, slope), axis=-1) # MEDIASTINAL window\n\n            img = self.transform(np.concatenate([image1, image2, image3], axis=-1)).astype(np.float32)\n        \n            return {'img': self._preprocess_img(img),\n                    'category': data.SOPInstanceUID}\n        except Exception as e:\n            return {'img': self._preprocess_img(np.zeros(512, 512, 3)),\n                    'category': data.SOPInstanceUID}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = ClassificationDataset()\nloader = DataLoader(dataset=test, batch_size=16, shuffle=False, num_workers=16, drop_last=False)\nsub = open('submission.csv', \"w\")\nsub.write('id,label')\nsub.write(\"\\n\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with torch.no_grad():\n    for batch in tqdm(loader):\n        imgs = batch['img']#.cuda()\n        names = batch['category']\n        y_preds = [0.5]*16#model(imgs).data.cpu()\n        for idx, name in enumerate(names):\n            sub.write(f'{name},{np.format_float_positional(y_preds[idx], precision=10)}')\n            sub.write(\"\\n\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv('../input/rsna-str-pulmonary-embolism-detection/test.csv')\nlabels = ['negative_exam_for_pe',\n               'rv_lv_ratio_gte_1',\n               'rv_lv_ratio_lt_1',\n               'leftsided_pe',\n               'chronic_pe',\n               'rightsided_pe',\n               'acute_and_chronic_pe',\n               'central_pe',\n               'indeterminate']\n\nfor study in tqdm(np.unique(test_df.StudyInstanceUID)):\n    for l in labels:\n        sub.write(f'{study}_{l},{0.5}')\n        sub.write(\"\\n\")\nsub.close()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# testDataDF = pd.read_csv('../input/rsna-str-pulmonary-embolism-detection/test.csv', dtype={'StudyInstanceUID':'string', 'SeriesInstanceUID':'string', 'SOPInstanceUID':'string'})\n# testDataDF = testDataDF.set_index('SOPInstanceUID')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# listOfStudyID = testDataDF['StudyInstanceUID'].unique()\n# print(len(listOfStudyID))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sanity Check\n#thisStudyDF.head()\n#print(len(thisStudyDF))\n\n#thisImageIDlist = thisStudyDF.index.to_list()\n#for eachItem in thisStudyDF.index:\n#    print(type(eachItem))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# def window(img, WL=50, WW=350):\n#     upper, lower = WL+WW//2, WL-WW//2\n#     X = np.clip(img.copy(), lower, upper)\n#     X = X - np.min(X)\n#     X = X / np.max(X)\n#     X = (X*255.0).astype('uint8')\n#     return X\n\n# data_transform = transforms.Compose([\n#         transforms.ToTensor(),\n#         transforms.Normalize(mean=[0.485, 0.456, 0.406],\n#                              std=[0.229, 0.224, 0.225])\n#     ])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nmodel_Path = '../input/firstbaselinemodel/baseMod4.pth' \nbaseModel = torch.load(model_Path) \nbaseModel.eval();\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# #scoreDF = pd.DataFrame(columns=['id','label'])\n# #scoreDF = scoreDF.set_index('id')\n\n# f = open('submission.csv', 'w')\n# f.write('id,label\\n')\n\n# with torch.no_grad():\n\n#     for eachStudyID in tqdm(listOfStudyID):\n        \n#         thisStudyDF = testDataDF[testDataDF['StudyInstanceUID']==eachStudyID]\n        \n#         for eachImageID in thisStudyDF.index:\n            \n#             '''\n#             try:\n#                 eachImagePath = '../input/rsna-str-pulmonary-embolism-detection/test/'+testDataDF.loc[eachImageID, 'StudyInstanceUID']+'/'+testDataDF.loc[eachImageID, 'SeriesInstanceUID']+'/'+eachImageID+'.dcm'\n#                 dcm_data = dcmread(eachImagePath)\n#                 image = dcm_data.pixel_array * int(dcm_data.RescaleSlope) + int(dcm_data.RescaleIntercept)\n#                 image = np.stack([window(image, WL=-600, WW=1500),\n#                                   window(image, WL=40, WW=400),\n#                                   window(image, WL=100, WW=700)], 2)\n\n#                 image = image.astype(np.float32)\n#                 image = data_transform(image)\n#                 toPred = image.unsqueeze(0).cuda()\n#                 z = baseModel(toPred)\n#                 pred = torch.sigmoid(z)\n#                 pred = pred.cpu().detach().numpy().astype('float32')[0,0]\n#             except:\n#                 pred = defaultScore['_pe_present_on_image']\n#             '''\n            \n#             #scoreDF.loc[imageID, 'label'] = 0.5\n#             f.write(eachImageID+',0.5\\n')\n            \n#         # Study level labels\n#         listOfMetricLabels = ['_negative_exam_for_pe', '_rv_lv_ratio_gte_1', '_rv_lv_ratio_lt_1', '_leftsided_pe', '_chronic_pe', '_rightsided_pe', '_acute_and_chronic_pe', '_central_pe', '_indeterminate']\n\n#         for eachMetric in listOfMetricLabels:\n#             #scoreDF.loc[studyID+eachMetric, 'label'] = 0.5\n#             f.write(eachStudyID+eachMetric+',0.5\\n')\n            \n# f.close()\n\n# #print(\"totalEntries\",len(scoreDF))\n# #scoreDF.to_csv('submission.csv', index=True)\n\n# print('finish')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# submissionDF = pd.read_csv('submission.csv', dtype={'id':'string', 'label':'string'})\n# submissionDF['label'].values\n# print(len(submissionDF))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}