{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Breast Baseline - Inference\n\nThis notebook shows a simple inference pipeline based on the pregenerated datasets.","metadata":{}},{"cell_type":"code","source":"DEBUG = False","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:24.734083Z","iopub.execute_input":"2022-12-05T04:56:24.734833Z","iopub.status.idle":"2022-12-05T04:56:24.740521Z","shell.execute_reply.started":"2022-12-05T04:56:24.734791Z","shell.execute_reply":"2022-12-05T04:56:24.738934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initialization","metadata":{}},{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-05T04:56:24.742542Z","iopub.execute_input":"2022-12-05T04:56:24.743529Z","iopub.status.idle":"2022-12-05T04:56:24.750986Z","shell.execute_reply.started":"2022-12-05T04:56:24.743492Z","shell.execute_reply":"2022-12-05T04:56:24.750078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport glob\nimport gdcm\nimport json\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-05T04:56:24.752707Z","iopub.execute_input":"2022-12-05T04:56:24.75383Z","iopub.status.idle":"2022-12-05T04:56:24.761531Z","shell.execute_reply.started":"2022-12-05T04:56:24.753717Z","shell.execute_reply":"2022-12-05T04:56:24.760592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation\n\n- I use the same strategy as in https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg","metadata":{}},{"cell_type":"code","source":"test_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\nif DEBUG:\n    test_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/10042/*.dcm\")\n    \nprint(\"Number of images :\", len(test_images))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:24.765453Z","iopub.execute_input":"2022-12-05T04:56:24.766498Z","iopub.status.idle":"2022-12-05T04:56:24.778954Z","shell.execute_reply.started":"2022-12-05T04:56:24.766455Z","shell.execute_reply":"2022-12-05T04:56:24.777872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_FOLDER = \"/kaggle/tmp/output/\"\nSIZE = 512\nEXTENSION = \"png\"\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:24.780443Z","iopub.execute_input":"2022-12-05T04:56:24.781272Z","iopub.status.idle":"2022-12-05T04:56:24.786777Z","shell.execute_reply.started":"2022-12-05T04:56:24.781227Z","shell.execute_reply":"2022-12-05T04:56:24.785822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n\n    cv2.imwrite(save_folder + f\"{patient}_{image}.{extension}\", (img * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:24.78822Z","iopub.execute_input":"2022-12-05T04:56:24.788863Z","iopub.status.idle":"2022-12-05T04:56:24.797988Z","shell.execute_reply.started":"2022-12-05T04:56:24.788825Z","shell.execute_reply":"2022-12-05T04:56:24.797151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(test_images)\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:24.800632Z","iopub.execute_input":"2022-12-05T04:56:24.800928Z","iopub.status.idle":"2022-12-05T04:56:30.806238Z","shell.execute_reply.started":"2022-12-05T04:56:24.800904Z","shell.execute_reply":"2022-12-05T04:56:30.805009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"markdown","source":"### Utils","metadata":{}},{"cell_type":"markdown","source":"### Main","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf['cancer'] = 0\ndf['path'] = SAVE_FOLDER + df[\"patient_id\"].astype(str) + \"_\" + df[\"image_id\"].astype(str) + \".png\"","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:30.813573Z","iopub.execute_input":"2022-12-05T04:56:30.815952Z","iopub.status.idle":"2022-12-05T04:56:30.841257Z","shell.execute_reply.started":"2022-12-05T04:56:30.815907Z","shell.execute_reply":"2022-12-05T04:56:30.840627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:30.842418Z","iopub.execute_input":"2022-12-05T04:56:30.842769Z","iopub.status.idle":"2022-12-05T04:56:30.861418Z","shell.execute_reply.started":"2022-12-05T04:56:30.842734Z","shell.execute_reply":"2022-12-05T04:56:30.860333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EXP_FOLDERS = [\n    \"/kaggle/input/rsna-breast-weights-public/\"\n]","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:30.862765Z","iopub.execute_input":"2022-12-05T04:56:30.863583Z","iopub.status.idle":"2022-12-05T04:56:30.867618Z","shell.execute_reply.started":"2022-12-05T04:56:30.863544Z","shell.execute_reply":"2022-12-05T04:56:30.8667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Main loop","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import transforms as T\n\nto_tensor = T.Compose([\n    T.ToTensor(),\n    T.Resize((256, 256)),\n    T.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n#     T.Normalize(mean=[0.66437738, 0.50478148, 0.70114894], std=[0.15825711, 0.24371008, 0.13832686])\n])\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:30.868771Z","iopub.execute_input":"2022-12-05T04:56:30.869599Z","iopub.status.idle":"2022-12-05T04:56:32.521414Z","shell.execute_reply.started":"2022-12-05T04:56:30.869562Z","shell.execute_reply":"2022-12-05T04:56:32.520437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import yaml\nfrom torchvision import models\nimport torch\nimport torch.nn as nn\nfrom torchvision import transforms as T\nfrom PIL import Image\nBACKBONE_MAPPING = {\n    'b0': {\n        'm': models.efficientnet_b0,\n        'last_layer': 'classifier',\n        'in_features': 1280\n    },\n    'b1': {\n        'm': models.efficientnet_b0,\n        'last_layer': 'classifier',\n        'in_features': 1280\n    },\n    'resnet18': {\n        'm': models.resnet18,\n        'last_layer': 'fc',\n        'in_features': 512\n    },\n    'resnet50': {\n        'm': models.resnet50,\n        'last_layer': 'fc',\n        'in_features': 2048\n    }\n}\n\ndef get_module_list(config):\n    model_list = []\n    for name, arg_list in config:\n        template = getattr(nn, name)\n        m = template(*arg_list)\n        model_list.append(m)\n    return model_list\n\nclass HEMaxBlock(nn.Module):\n    def __init__(self, beta):\n        super(HEMaxBlock, self).__init__()\n        self.beta = beta\n    \n    def forward(self, X):\n        '''\n            X: (batch_size, channel, w, h)\n        '''\n        shape = X.shape\n        \n        X_mask = X.clone()\n        max_value, _ = torch.max(X_mask.view(*shape[:2], -1), dim = -1)\n        mask = X_mask == max_value[:, :, None, None]\n        \n        one = torch.ones_like(X)\n        one[mask] *= self.beta\n        X = X * one\n        return X\n\nclass DynamicHE(nn.Module):\n    def __init__(self, backbone, cut_layer, bottle_layers, head_layers, beta):\n        super().__init__()\n        net_info = BACKBONE_MAPPING.get(backbone)\n        net = net_info['m'](pretrained=False)\n        blocks = []\n        for name, layer in net.named_children():\n            blocks.append(layer)\n            if name == cut_layer[0]:\n                self.block_pre = nn.Sequential(*blocks)\n                blocks = []\n            if name == cut_layer[1]:\n                self.block_pos = nn.Sequential(*blocks)\n                break\n        self.bottle_layers = nn.Sequential(*get_module_list(bottle_layers))\n        self.head = nn.ModuleList(\n            nn.Sequential(*get_module_list(head))\n            for head in head_layers\n        )\n        self.n_head = len(head_layers)\n        self.he_block = HEMaxBlock(beta)\n\n    def forward(self, X):\n        '''\n            X: (batch_size, channel, w, h)\n        '''\n        out = self.block_pre(X)\n        if self.training:\n            out = self.he_block(out)\n        out = self.block_pos(out)\n        out = nn.functional.adaptive_avg_pool2d(out, (1, 1))\n        out = out.flatten(1)\n        out = self.bottle_layers(out)\n        out_head = []\n        for i in range(self.n_head):\n            out_head.append(self.head[i](out))\n\n        return out_head\n    \n\n\nclass Cls:\n    def __init__(self, model_cf):\n        self.device = model_cf['device']\n        self.model = DynamicHE(**model_cf['args'])\n        self.model.to(device=self.device)\n        checkpoint = torch.load(model_cf['weight'], map_location=self.device)\n        self.model.load_state_dict(checkpoint['model'])\n        self.model.eval()\n\n        self.to_tensor = T.Compose([\n                                    T.ToTensor(),\n                                    T.Resize((256,256)),\n                                    T.Normalize(mean=[0.485, 0.456, 0.406], \n                                                std=[0.229, 0.224, 0.225])\n                                ])\n    def __call__(self, imgs):\n        imgs = torch.stack([self.to_tensor(img) for img in imgs]).to(device = self.device)\n        with torch.no_grad():\n            outputs = self.model(imgs)\n        \n        pred_label = [torch.softmax(output, dim = 1).cpu().tolist() for output in outputs]\n        \n        return [np.argmax(p).item() for p in pred_label[0]]\n        ","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:32.522739Z","iopub.execute_input":"2022-12-05T04:56:32.524175Z","iopub.status.idle":"2022-12-05T04:56:32.56284Z","shell.execute_reply.started":"2022-12-05T04:56:32.524137Z","shell.execute_reply":"2022-12-05T04:56:32.561993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config={'weight': '/kaggle/input/weights/resnet50_cut256_2cls.pt', \n        'device': 'cuda', \n        'args': {\n            'backbone': 'resnet50', \n            'beta': 0.5, \n            'bottle_layers': [\n                ['Linear', [2048, 512]], \n                ['BatchNorm1d', [512]],\n                ['ReLU', []],\n                ['Dropout', [0.5]]\n            ], \n            'cut_layer': ['layer4', 'layer4'], \n            'head_layers': [\n                [['Linear', [512, 2]]], # cancer\n                #[['Linear', [512, 2]]], # laterality\n                #[['Linear', [512, 2]]], # view\n                #[['Linear', [512, 2]]], # invasive\n                #[['Linear', [512, 3]]], # BIRADS\n                #[['Linear', [512, 2]]], # implant\n                [['Linear', [512, 4]]]] # density\n        } \n       }","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:32.564063Z","iopub.execute_input":"2022-12-05T04:56:32.564504Z","iopub.status.idle":"2022-12-05T04:56:32.57117Z","shell.execute_reply.started":"2022-12-05T04:56:32.564469Z","shell.execute_reply":"2022-12-05T04:56:32.570004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Cls(config)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:32.572753Z","iopub.execute_input":"2022-12-05T04:56:32.5731Z","iopub.status.idle":"2022-12-05T04:56:38.581459Z","shell.execute_reply.started":"2022-12-05T04:56:32.573059Z","shell.execute_reply":"2022-12-05T04:56:38.580427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = df.path.to_list()\nlaterality = df.laterality.to_list()\nbatch_size = 32\npred = []\nfor idx in range(0, len(path), batch_size):\n#     imgs = [Image.open(p).convert('RGB') for p in path[idx:idx + batch_size]]\n    imgs = []\n    for p, l in zip(path[idx:idx + batch_size], laterality[idx:idx + batch_size]):\n        img = Image.open(p).convert('RGB')\n        width, height= img.size\n\n        if l == 'L':\n            left = 0\n            right = int(width / 3 * 2)\n        else:\n            left = int(width / 3)\n            right = width\n        croped_img = img.crop((left, 0, right, height))\n        imgs.append(croped_img)\n\n    out = model(imgs)\n    pred.extend(out)    ","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:56:58.236352Z","iopub.execute_input":"2022-12-05T04:56:58.236745Z","iopub.status.idle":"2022-12-05T04:57:03.809124Z","shell.execute_reply.started":"2022-12-05T04:56:58.236714Z","shell.execute_reply":"2022-12-05T04:57:03.807878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred, path","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:57:04.716965Z","iopub.execute_input":"2022-12-05T04:57:04.717325Z","iopub.status.idle":"2022-12-05T04:57:04.724346Z","shell.execute_reply.started":"2022-12-05T04:57:04.717294Z","shell.execute_reply":"2022-12-05T04:57:04.723396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub = pd.DataFrame({\n#     'prediction_id': df.prediction_id,\n#     'cancer': pred\n# })\n\ndf['cancer'] = pred\n\nsub = df[['prediction_id', 'cancer']].groupby(\"prediction_id\").mean().reset_index()\n# sub.to_csv('submission.csv', index=False)\nsub.to_csv('/kaggle/working/submission.csv', index=False)\n\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T04:57:13.041199Z","iopub.execute_input":"2022-12-05T04:57:13.041628Z","iopub.status.idle":"2022-12-05T04:57:13.066285Z","shell.execute_reply.started":"2022-12-05T04:57:13.041591Z","shell.execute_reply":"2022-12-05T04:57:13.06542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Done !","metadata":{}}]}