{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport torch \nimport torch.nn as nn \n\nimport os\nfrom torchvision.models.efficientnet import efficientnet_b0 \nfrom torchvision.transforms import transforms\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data import Dataset \nfrom tqdm import tqdm\nimport glob\nfrom PIL import Image\n\nimport matplotlib.pyplot as plt\nimport cv2\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-19T05:58:30.113243Z","iopub.execute_input":"2023-02-19T05:58:30.113729Z","iopub.status.idle":"2023-02-19T05:58:32.096692Z","shell.execute_reply.started":"2023-02-19T05:58:30.113649Z","shell.execute_reply":"2023-02-19T05:58:32.095601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import PIL\nimport numpy as np\nimport random \nimport torch \n\nclass RSNABreast(Dataset):\n    def __init__(self, dataframe,  transform = None):\n        self.dataframe = dataframe.reset_index(drop=True)\n        self.transform = transform \n        #self.cancerlist = self.dataframe[self.dataframe['cancer'] == 1].index \n        \n    def __len__(self):\n        return len(self.dataframe)\n    \n    def __getitem__(self,idx):\n        #if random.uniform(0,1) > 0.8:\n            #idx = self.cancerlist[np.random.randint(len(self.cancerlist))] \n        img = self.load_image(idx)\n        val = self.dataframe.iloc[idx]\n        fname = str(val['patient_id']) + '_' + str(val['laterality'])\n        #class_name = self.load_class(idx)\n        #print(class_name)\n        if self.transform:\n            img = self.transform(img)\n        return img, fname #torch.tensor(class_name, dtype=(torch.float)) \n    \n    def load_image(self,image_index):\n        #MLO_image = pydicom.dcmread(MLO_image_path).pixel_array\n        #dicomim = dicom.read_file(self.dataframe['Filepath'][image_index])\n        #data = dicomim.pixel_array\n        #if dicomim.PhotometricInterpretation == \"MONOCHROME1\":\n            #data = np.amax(data) - data\n        #data = data - np.min(data)\n        #data = data / np.max(data)\n        #data = (data * 255).astype(np.uint8)    \n        #new_image = data.astype(float)\n        #final_image = Image.fromarray(new_image)\n        #img = final_image.convert(\"RGB\")\n        img = PIL.Image.open(self.dataframe['Filepath'][image_index])\n        img  = img.convert(\"RGB\")\n        #print(self.file_path_list[image_index]) \n        return img\n    \n    def load_class(self, image_index):\n        class_row = self.dataframe.iloc[image_index]\n        class_name = class_row['cancer']\n        #print(class_name)\n        return class_name","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:58:32.102537Z","iopub.execute_input":"2023-02-19T05:58:32.105405Z","iopub.status.idle":"2023-02-19T05:58:32.12053Z","shell.execute_reply.started":"2023-02-19T05:58:32.105364Z","shell.execute_reply":"2023-02-19T05:58:32.119549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n    !mkdir -p /root/.cache/torch/hub/checkpoints/\n    !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n    !pip install /kaggle/input/rsna-2022-whl/{torch-1.12.1-cp37-cp37m-manylinux1_x86_64.whl,torchvision-0.13.1-cp37-cp37m-manylinux1_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:58:32.124878Z","iopub.execute_input":"2023-02-19T05:58:32.1255Z","iopub.status.idle":"2023-02-19T05:59:25.979334Z","shell.execute_reply.started":"2023-02-19T05:58:32.125463Z","shell.execute_reply":"2023-02-19T05:59:25.97777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\n#TAKEN FROM : https://www.kaggle.com/code/vslaykovsky/infer-effnetv2-aux-targets-weighted-loss-thres/notebook\nfrom joblib import Parallel, delayed\n\ndef process(f, size=256, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n    \n    minX =  img.min()\n    maxX = img.max()\n\n    img = img[:,img.std(axis=0) > 60]\n    img = img[img.std(axis=1) > 60, :]\n    \n\n    img = (img - minX) / (maxX - minX)\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n        \n    X = img[5:-5, 5:-5]\n    \n    output= cv2.connectedComponentsWithStats((X > 0.05).astype(np.uint8)[:, :], 8, cv2.CV_32S)\n\n    stats = output[2]\n    #img = cv2.resize(img, (512, 1024))\n    idx = stats[1:, 4].argmax() + 1\n    x1, y1, w, h = stats[idx][:4]\n    x2 = x1 + w\n    y2 = y1 + h\n    \n    # cutting out the breast data\n    #try:\n    X_fit = X[y1: y2, x1: x2]\n    X_fit = cv2.resize(X_fit, (512, 1024))\n    cv2.imwrite(save_folder + f\"/{patient}_{image}.{extension}\", (X_fit * 255).astype(np.uint8))\n    \n\n\n\ntest_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\n!mkdir -p test\n_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, save_folder='test')\n    for uid in tqdm(test_images)\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:59:25.986413Z","iopub.execute_input":"2023-02-19T05:59:25.98893Z","iopub.status.idle":"2023-02-19T05:59:35.664974Z","shell.execute_reply.started":"2023-02-19T05:59:25.988886Z","shell.execute_reply":"2023-02-19T05:59:35.663639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weightPath = '/kaggle/input/rsnabreastscancer/EffiB0_oversample4/RSNABC.pt'\ndevice = \"cuda:0\"\n\nmodel  = torch.load('/kaggle/input/rsnabreastscancer/norm.pt', map_location=torch.device('cpu'))\n#model = efficientnet_b0(False,False); \nmodel.classifier[1] = nn.Linear(1280,1,bias=True) \nmodel.load_state_dict(torch.load(weightPath,map_location='cpu'))","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:59:35.671069Z","iopub.execute_input":"2023-02-19T05:59:35.673369Z","iopub.status.idle":"2023-02-19T05:59:36.223947Z","shell.execute_reply.started":"2023-02-19T05:59:35.673322Z","shell.execute_reply":"2023-02-19T05:59:36.223038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = '/kaggle/working/test'\n\ndef getpath(root_path, parent_folder, row_df):\n    path = os.path.join(root_path, str(row_df['patient_id']) + '_' +str(row_df['image_id']) + '.png')\n    return path","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:59:36.22541Z","iopub.execute_input":"2023-02-19T05:59:36.225773Z","iopub.status.idle":"2023-02-19T05:59:36.233914Z","shell.execute_reply.started":"2023-02-19T05:59:36.225738Z","shell.execute_reply":"2023-02-19T05:59:36.232698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_sub = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:59:36.236431Z","iopub.execute_input":"2023-02-19T05:59:36.236711Z","iopub.status.idle":"2023-02-19T05:59:36.253487Z","shell.execute_reply.started":"2023-02-19T05:59:36.236686Z","shell.execute_reply":"2023-02-19T05:59:36.25267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\npareF = 'test_images' \n\ntest_csv['Filepath']  = test_csv.apply(lambda row : getpath(root_path, pareF, row), axis=1)\n\ntest_csv.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-19T05:59:36.256357Z","iopub.execute_input":"2023-02-19T05:59:36.256611Z","iopub.status.idle":"2023-02-19T05:59:36.282071Z","shell.execute_reply.started":"2023-02-19T05:59:36.256587Z","shell.execute_reply":"2023-02-19T05:59:36.281213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_transform = transforms.Compose([transforms.Resize([512,256]),transforms.ToTensor(),\n                                      transforms.Normalize(( [0.485, 0.456, 0.406] ), \n                                                           ([0.229, 0.224, 0.225]))]) \n\nds_loader= RSNABreast(test_csv, valid_transform)\ndl_l = DataLoader(ds_loader,32)\ndevice = torch.device(\"cuda:0\")\nmodel = model.to(device)\n\n\nlist_prediction_id = [] \nlist_cancer_score = [] \nfor xb , yb in tqdm(dl_l):\n    xb = xb.to(device)\n    #yb = yb.to(device)\n    with torch.no_grad():\n        outM  = model(xb)\n        sigV  = nn.Sigmoid()(outM).to(\"cpu\")\n    \n    for indx, val in enumerate(yb):\n        list_prediction_id.append(yb[indx])\n        list_cancer_score.append(sigV[indx].item())","metadata":{"execution":{"iopub.status.busy":"2023-02-19T06:00:18.152575Z","iopub.execute_input":"2023-02-19T06:00:18.152984Z","iopub.status.idle":"2023-02-19T06:00:18.235887Z","shell.execute_reply.started":"2023-02-19T06:00:18.152947Z","shell.execute_reply":"2023-02-19T06:00:18.234809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"input_tensor.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-04T15:22:55.604542Z","iopub.execute_input":"2022-12-04T15:22:55.605105Z","iopub.status.idle":"2022-12-04T15:22:55.61167Z","shell.execute_reply.started":"2022-12-04T15:22:55.605078Z","shell.execute_reply":"2022-12-04T15:22:55.610719Z"}}},{"cell_type":"code","source":"list_cancer_score","metadata":{"execution":{"iopub.status.busy":"2023-02-19T06:00:19.980597Z","iopub.execute_input":"2023-02-19T06:00:19.980961Z","iopub.status.idle":"2023-02-19T06:00:19.987509Z","shell.execute_reply.started":"2023-02-19T06:00:19.980931Z","shell.execute_reply":"2023-02-19T06:00:19.986465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_sub = pd.DataFrame()\nsamp_sub['prediction_id'] = list_prediction_id\nsamp_sub['cancer'] = list_cancer_score ","metadata":{"execution":{"iopub.status.busy":"2023-02-19T06:00:23.538559Z","iopub.execute_input":"2023-02-19T06:00:23.538923Z","iopub.status.idle":"2023-02-19T06:00:23.546378Z","shell.execute_reply.started":"2023-02-19T06:00:23.538891Z","shell.execute_reply":"2023-02-19T06:00:23.545288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_sub\nsamp_sub = samp_sub.sort_values(by='cancer', ascending=False)\nsamp_sub = samp_sub.drop_duplicates(subset='prediction_id', keep=\"first\")\nsamp_sub.to_csv('submission.csv', index= False)\nsamp_sub","metadata":{"execution":{"iopub.status.busy":"2023-02-19T06:00:23.800982Z","iopub.execute_input":"2023-02-19T06:00:23.801555Z","iopub.status.idle":"2023-02-19T06:00:23.815991Z","shell.execute_reply.started":"2023-02-19T06:00:23.801515Z","shell.execute_reply":"2023-02-19T06:00:23.814893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}