{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2023-09-28T17:36:34.149775Z","iopub.execute_input":"2023-09-28T17:36:34.150223Z","iopub.status.idle":"2023-09-28T17:36:54.911078Z","shell.execute_reply.started":"2023-09-28T17:36:34.150187Z","shell.execute_reply":"2023-09-28T17:36:54.909355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport glob\nimport gdcm\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport json\nfrom PIL import Image\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\nimport torch\nimport torch.nn as nn\nfrom torch import Tensor\nfrom typing import Type\nimport torchvision\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-09-28T17:37:09.742436Z","iopub.execute_input":"2023-09-28T17:37:09.743826Z","iopub.status.idle":"2023-09-28T17:37:14.171877Z","shell.execute_reply.started":"2023-09-28T17:37:09.743759Z","shell.execute_reply":"2023-09-28T17:37:14.170049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_path = \"/kaggle/input/rsna-breast-cancer-256-pngs/\"\ntrain_df['path'] = train_path+train_df.patient_id.astype(str)+'_'+train_df.image_id.astype(str)+'.png'\ntrain_df['path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-09-28T17:37:14.174183Z","iopub.execute_input":"2023-09-28T17:37:14.174946Z","iopub.status.idle":"2023-09-28T17:37:14.535034Z","shell.execute_reply.started":"2023-09-28T17:37:14.174897Z","shell.execute_reply":"2023-09-28T17:37:14.533348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['transform'] = False","metadata":{"execution":{"iopub.status.busy":"2023-09-28T17:38:09.830366Z","iopub.execute_input":"2023-09-28T17:38:09.830837Z","iopub.status.idle":"2023-09-28T17:38:09.838627Z","shell.execute_reply.started":"2023-09-28T17:38:09.830797Z","shell.execute_reply":"2023-09-28T17:38:09.836933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore the dataset","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:06:41.203733Z","iopub.execute_input":"2023-04-02T20:06:41.2042Z","iopub.status.idle":"2023-04-02T20:06:41.214902Z","shell.execute_reply.started":"2023-04-02T20:06:41.204162Z","shell.execute_reply":"2023-04-02T20:06:41.213241Z"}}},{"cell_type":"code","source":"num_cancerous = len(train_df[train_df['cancer'] == 1])\nnum_not_cancerous = len(train_df) - num_cancerous\n\ndata = [num_cancerous, num_not_cancerous]\n\nlabels = ['Cancerous Examples (%s)' %  (num_cancerous), 'Non-Cancerous Examples (%s)' % (num_not_cancerous)]\n\ncolors = sns.color_palette('pastel')[0:2]\nplt.pie(data, labels=labels, colors=colors, autopct='%.0f%%')\nplt.title(\"Cancerous vs Non Cancerous Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:34:52.578016Z","iopub.execute_input":"2023-04-02T20:34:52.578478Z","iopub.status.idle":"2023-04-02T20:34:52.744157Z","shell.execute_reply.started":"2023-04-02T20:34:52.57844Z","shell.execute_reply":"2023-04-02T20:34:52.742334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cancerous = len(train_df[train_df['cancer'] == 1]) * 11\nnum_not_cancerous = len(train_df) - num_cancerous\n\ndata = [num_cancerous, num_not_cancerous]\n\nlabels = ['Cancerous Examples (%s)' %  (num_cancerous), 'Non-Cancerous Examples (%s)' % (num_not_cancerous)]\n\ncolors = sns.color_palette('pastel')[0:2]\nplt.pie(data, labels=labels, colors=colors, autopct='%.0f%%')\nplt.title(\"Cancerous vs Non Cancerous Examples After Augmentation Replication\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:39:22.210893Z","iopub.execute_input":"2023-04-02T20:39:22.211319Z","iopub.status.idle":"2023-04-02T20:39:22.377671Z","shell.execute_reply.started":"2023-04-02T20:39:22.211285Z","shell.execute_reply":"2023-04-02T20:39:22.375904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Making copies of Cancerous Images","metadata":{}},{"cell_type":"code","source":"cancerous = train_df[train_df['cancer'] == 1]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:34:54.554569Z","iopub.execute_input":"2023-04-02T20:34:54.555936Z","iopub.status.idle":"2023-04-02T20:34:54.563312Z","shell.execute_reply.started":"2023-04-02T20:34:54.555886Z","shell.execute_reply":"2023-04-02T20:34:54.562098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def augment_dup_cancerous_imgs(num_duplicates=10, cancerous=cancerous):\n    temp_df = cancerous\n    for i, row in cancerous.iterrows():\n        for j in range(num_duplicates):\n            temp = row\n            temp['transform'] = True\n            temp_df = temp_df.append(temp, ignore_index=True)\n            \n    return temp_df           ","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:34:55.254991Z","iopub.execute_input":"2023-04-02T20:34:55.255443Z","iopub.status.idle":"2023-04-02T20:34:55.262415Z","shell.execute_reply.started":"2023-04-02T20:34:55.255405Z","shell.execute_reply":"2023-04-02T20:34:55.26144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_with_dups = augment_dup_cancerous_imgs()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:34:55.964956Z","iopub.execute_input":"2023-04-02T20:34:55.965638Z","iopub.status.idle":"2023-04-02T20:35:59.309807Z","shell.execute_reply.started":"2023-04-02T20:34:55.9656Z","shell.execute_reply":"2023-04-02T20:35:59.308301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(cancer_with_dups[cancer_with_dups['transform'] == False]), len(cancer_with_dups[cancer_with_dups['transform'] == True]))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T01:43:25.419651Z","iopub.execute_input":"2023-04-02T01:43:25.420259Z","iopub.status.idle":"2023-04-02T01:43:25.438348Z","shell.execute_reply.started":"2023-04-02T01:43:25.420203Z","shell.execute_reply":"2023-04-02T01:43:25.436742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Append cancerous duplicate df to train_df","metadata":{}},{"cell_type":"code","source":"print(len(train_df))\ntrain_with_dups = train_df.append(cancer_with_dups)\nshuffled_train = train_with_dups.sample(frac=1).reset_index()\nprint(\"RAN\")\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:37:28.896414Z","iopub.execute_input":"2023-04-02T20:37:28.897616Z","iopub.status.idle":"2023-04-02T20:37:28.959025Z","shell.execute_reply.started":"2023-04-02T20:37:28.897553Z","shell.execute_reply":"2023-04-02T20:37:28.957835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shuffled_train = train_with_dups.sample(frac=1)\nshuffled_train.reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:37:31.213261Z","iopub.execute_input":"2023-04-02T20:37:31.213702Z","iopub.status.idle":"2023-04-02T20:37:31.277129Z","shell.execute_reply.started":"2023-04-02T20:37:31.213665Z","shell.execute_reply":"2023-04-02T20:37:31.275804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Undersampling non-cancerous examples (only run if you mean to do this)","metadata":{}},{"cell_type":"code","source":"# taking every cancerous example we have and undersampling the non cancerous for a 50-50 split\nprint(len(shuffled_train[shuffled_train['cancer'] == True]))\nundersampled_df = pd.DataFrame()\nundersampled_df = undersampled_df.append(shuffled_train[shuffled_train['cancer'] == True].sample(13896))\nundersampled_df = undersampled_df.append(shuffled_train[shuffled_train['cancer'] == False].sample(13896))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:37:33.743838Z","iopub.execute_input":"2023-04-02T20:37:33.744274Z","iopub.status.idle":"2023-04-02T20:37:33.787387Z","shell.execute_reply.started":"2023-04-02T20:37:33.744238Z","shell.execute_reply":"2023-04-02T20:37:33.786039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(undersampled_df[undersampled_df['cancer'] == True]))\n\nshuffled_train = undersampled_df\nshuffled_train = shuffled_train.reset_index(drop=True)\n# print(len(shuffled_train[shuffled_train['cancer'] == False]))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:37:38.07318Z","iopub.execute_input":"2023-04-02T20:37:38.073585Z","iopub.status.idle":"2023-04-02T20:37:38.091965Z","shell.execute_reply.started":"2023-04-02T20:37:38.073551Z","shell.execute_reply":"2023-04-02T20:37:38.090614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"13896 - 12738","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:38:43.987439Z","iopub.execute_input":"2023-04-02T20:38:43.988739Z","iopub.status.idle":"2023-04-02T20:38:43.994899Z","shell.execute_reply.started":"2023-04-02T20:38:43.988692Z","shell.execute_reply":"2023-04-02T20:38:43.994017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cancerous = len(undersampled_df[undersampled_df['cancer'] == 1]) \nnum_not_cancerous = len(undersampled_df) - num_cancerous\n\ndata = [num_cancerous, num_not_cancerous]\n\nlabels = ['Cancerous Examples 12738', 'Non-Cancerous Examples 12738']\n\ncolors = sns.color_palette('pastel')[0:2]\nplt.pie(data, labels=labels, colors=colors, autopct='%.0f%%')\nplt.title(\"Cancerous vs Non Cancerous Examples After Undersampling\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:40:39.949992Z","iopub.execute_input":"2023-04-02T20:40:39.950491Z","iopub.status.idle":"2023-04-02T20:40:40.087946Z","shell.execute_reply.started":"2023-04-02T20:40:39.950435Z","shell.execute_reply":"2023-04-02T20:40:40.086496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(shuffled_train))\nprint(len(train_df))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:24:23.888532Z","iopub.execute_input":"2023-04-02T04:24:23.888988Z","iopub.status.idle":"2023-04-02T04:24:23.895583Z","shell.execute_reply.started":"2023-04-02T04:24:23.888951Z","shell.execute_reply":"2023-04-02T04:24:23.894472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PermuteDimensionsTransform():\n    def __call__(self, image):\n        image = image.permute(2,0,1)\n        return image\n\nclass PermuteDimensionsTransformBack():\n    def __call__(self, image):\n        image = image.permute(1,2,0)\n        return image\n    \n\nclass RSNADataset():\n    \n    def __init__(self, df):\n        self.df = df\n        self.transform = torchvision.transforms.Compose([\n            PermuteDimensionsTransform(),\n            torchvision.transforms.RandomRotation(degrees=(0,60)),\n            PermuteDimensionsTransformBack(),\n            torchvision.transforms.RandomHorizontalFlip(p=0.5),\n            torchvision.transforms.RandomVerticalFlip(p=0.5)\n        ])\n\n    def __len__(self):\n        return len(self.df)\n    \n    def getimginfo(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['path']\n        image = cv2.imread(img_path).astype(np.float32)/255\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        if row['transform'] == True:\n            image = np.array(self.transform(torch.tensor(image)))\n        target = torch.tensor(row.cancer).float()\n        \n        \n        return {\"image\": image, \"target\": target}","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:33:30.249748Z","iopub.execute_input":"2023-04-02T04:33:30.251279Z","iopub.status.idle":"2023-04-02T04:33:30.263411Z","shell.execute_reply.started":"2023-04-02T04:33:30.251217Z","shell.execute_reply":"2023-04-02T04:33:30.262031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add more stuff in this function like flags for different ways to show the image\n\ndef displayImage(img, rgb):\n\n    plt.imshow(img)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:25:08.701711Z","iopub.execute_input":"2023-04-02T04:25:08.702599Z","iopub.status.idle":"2023-04-02T04:25:08.707786Z","shell.execute_reply.started":"2023-04-02T04:25:08.70254Z","shell.execute_reply":"2023-04-02T04:25:08.706859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shuffled_train[shuffled_train['transform'] == True]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:25:10.684539Z","iopub.execute_input":"2023-04-02T04:25:10.685037Z","iopub.status.idle":"2023-04-02T04:25:10.720921Z","shell.execute_reply.started":"2023-04-02T04:25:10.684991Z","shell.execute_reply":"2023-04-02T04:25:10.719517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataLoader = RSNADataset(shuffled_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:33:34.224288Z","iopub.execute_input":"2023-04-02T04:33:34.22475Z","iopub.status.idle":"2023-04-02T04:33:34.231203Z","shell.execute_reply.started":"2023-04-02T04:33:34.224712Z","shell.execute_reply":"2023-04-02T04:33:34.229715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_data = dataLoader.getimginfo(1)\ndisplayImage(img_data['image'], 1)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:32:55.188105Z","iopub.execute_input":"2023-04-02T04:32:55.188689Z","iopub.status.idle":"2023-04-02T04:32:55.485864Z","shell.execute_reply.started":"2023-04-02T04:32:55.188644Z","shell.execute_reply":"2023-04-02T04:32:55.484341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BasicBlock(nn.Module):\n    def __init__(\n        self, \n        in_channels: int,\n        out_channels: int,\n        stride: int = 1,\n        expansion: int = 1,\n        downsample: nn.Module = None\n    ) -> None:\n        super(BasicBlock, self).__init__()\n        # Multiplicative factor for the subsequent conv2d layer's output channels.\n        # It is 1 for ResNet18 and ResNet34.\n        self.expansion = expansion\n        self.downsample = downsample\n        self.conv1 = nn.Conv2d(\n            in_channels, \n            out_channels, \n            kernel_size=3, \n            stride=stride, \n            padding=1,\n            bias=False\n        )\n        self.bn1 = nn.BatchNorm2d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv2d(\n            out_channels, \n            out_channels*self.expansion, \n            kernel_size=3, \n            padding=1,\n            bias=False\n        )\n        self.bn2 = nn.BatchNorm2d(out_channels*self.expansion)\n    def forward(self, x: Tensor) -> Tensor:\n        identity = x\n        out = self.conv1(x)\n        out = self.bn1(out)\n        out = self.relu(out)\n        out = self.conv2(out)\n        out = self.bn2(out)\n        if self.downsample is not None:\n            identity = self.downsample(x)\n        out += identity\n        out = self.relu(out)\n        return  out","metadata":{"execution":{"iopub.status.busy":"2023-04-02T06:23:24.793166Z","iopub.execute_input":"2023-04-02T06:23:24.793664Z","iopub.status.idle":"2023-04-02T06:23:24.808792Z","shell.execute_reply.started":"2023-04-02T06:23:24.79362Z","shell.execute_reply":"2023-04-02T06:23:24.807025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNet(nn.Module):\n    def __init__(\n        self, \n        img_channels: int,\n        num_layers: int,\n        block: Type[BasicBlock],\n        num_classes: int  = 1000\n    ) -> None:\n        super(ResNet, self).__init__()\n        if num_layers == 18:\n            # The following `layers` list defines the number of `BasicBlock` \n            # to use to build the network and how many basic blocks to stack\n            # together.\n            layers = [2, 2, 2, 2]\n            self.expansion = 1\n        \n        self.in_channels = 64\n        # All ResNets (18 to 152) contain a Conv2d => BN => ReLU for the first\n        # three layers. Here, kernel size is 7.\n        self.conv1 = nn.Conv2d(\n            in_channels=img_channels,\n            out_channels=self.in_channels,\n            kernel_size=7, \n            stride=2,\n            padding=3,\n            bias=False\n        )\n        self.bn1 = nn.BatchNorm2d(self.in_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n\n        self.layer1 = self._make_layer(block, 64, layers[0])\n        self.layer2 = self._make_layer(block, 128, layers[1], stride=2)\n        self.layer3 = self._make_layer(block, 256, layers[2], stride=2)\n        self.layer4 = self._make_layer(block, 512, layers[3], stride=2)\n\n        self.avgpool = nn.AdaptiveAvgPool2d((1, 1))\n        self.fc = nn.Linear(512*self.expansion, num_classes)\n\n    def _make_layer(\n        self, \n        block: Type[BasicBlock],\n        out_channels: int,\n        blocks: int,\n        stride: int = 1\n    ) -> nn.Sequential:\n        downsample = None\n        if stride != 1:\n            \"\"\"\n            This should pass from `layer2` to `layer4` or \n            when building ResNets50 and above. Section 3.3 of the paper\n            Deep Residual Learning for Image Recognition\n            (https://arxiv.org/pdf/1512.03385v1.pdf).\n            \"\"\"\n            downsample = nn.Sequential(\n                nn.Conv2d(\n                    self.in_channels, \n                    out_channels*self.expansion,\n                    kernel_size=1,\n                    stride=stride,\n                    bias=False \n                ),\n                nn.BatchNorm2d(out_channels * self.expansion),\n            )\n        layers = []\n        layers.append(\n            block(\n                self.in_channels, out_channels, stride, self.expansion, downsample\n            )\n        )\n        self.in_channels = out_channels * self.expansion\n\n        for i in range(1, blocks):\n            layers.append(block(\n                self.in_channels,\n                out_channels,\n                expansion=self.expansion\n            ))\n        return nn.Sequential(*layers)\n\n    def forward(self, x: Tensor) -> Tensor:\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu(x)\n        x = self.maxpool(x)\n\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n        x = self.layer4(x)\n        # The spatial dimension of the final layer's feature \n        # map should be (7, 7) for all ResNets.\n     #   print('Dimensions of the last convolutional feature map: ', x.shape)\n\n        x = self.avgpool(x)\n        x = torch.flatten(x, 1)\n        x = self.fc(x)\n        x = self.relu(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-02T06:23:25.215351Z","iopub.execute_input":"2023-04-02T06:23:25.215813Z","iopub.status.idle":"2023-04-02T06:23:25.235266Z","shell.execute_reply.started":"2023-04-02T06:23:25.215773Z","shell.execute_reply":"2023-04-02T06:23:25.233753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(img_data['image']))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:20:27.406983Z","iopub.execute_input":"2023-04-02T04:20:27.407429Z","iopub.status.idle":"2023-04-02T04:20:27.414433Z","shell.execute_reply.started":"2023-04-02T04:20:27.407393Z","shell.execute_reply":"2023-04-02T04:20:27.412833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tensor = torch.Tensor(img_data['image'])\nprint(tensor.shape)\nmodel = ResNet(img_channels=256, num_layers=18, block=BasicBlock, num_classes=1)\nresult = model(tensor.unsqueeze(0))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T06:23:56.766041Z","iopub.execute_input":"2023-04-02T06:23:56.766553Z","iopub.status.idle":"2023-04-02T06:23:56.925808Z","shell.execute_reply.started":"2023-04-02T06:23:56.766507Z","shell.execute_reply":"2023-04-02T06:23:56.924328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(result, '\\n', img_data['target'])","metadata":{"execution":{"iopub.status.busy":"2023-04-02T06:23:58.435032Z","iopub.execute_input":"2023-04-02T06:23:58.435535Z","iopub.status.idle":"2023-04-02T06:23:58.443112Z","shell.execute_reply.started":"2023-04-02T06:23:58.435487Z","shell.execute_reply":"2023-04-02T06:23:58.442068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-04-01T20:02:58.967486Z","iopub.execute_input":"2023-04-01T20:02:58.967894Z","iopub.status.idle":"2023-04-01T20:02:58.998411Z","shell.execute_reply.started":"2023-04-01T20:02:58.967859Z","shell.execute_reply":"2023-04-01T20:02:58.997462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ny = shuffled_train['cancer']\nX = shuffled_train.drop(columns = ['cancer'])\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state=3)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:33:40.925873Z","iopub.execute_input":"2023-04-02T04:33:40.926352Z","iopub.status.idle":"2023-04-02T04:33:40.945772Z","shell.execute_reply.started":"2023-04-02T04:33:40.926313Z","shell.execute_reply":"2023-04-02T04:33:40.944342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(shuffled_train))\nprint(len(X_train), len(X_test), len(y_train), len(y_test))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:30:28.030625Z","iopub.execute_input":"2023-04-02T04:30:28.031096Z","iopub.status.idle":"2023-04-02T04:30:28.038591Z","shell.execute_reply.started":"2023-04-02T04:30:28.031055Z","shell.execute_reply":"2023-04-02T04:30:28.036855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(X_test[X_test['transform'] == True]))\nprint(len(X_train[X_train['transform'] == True]))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:23:00.393364Z","iopub.execute_input":"2023-04-02T04:23:00.393805Z","iopub.status.idle":"2023-04-02T04:23:00.406167Z","shell.execute_reply.started":"2023-04-02T04:23:00.393766Z","shell.execute_reply":"2023-04-02T04:23:00.404439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, row in X_train.iterrows():\n    print(idx)\n    break","metadata":{"execution":{"iopub.status.busy":"2023-04-02T01:45:21.318234Z","iopub.execute_input":"2023-04-02T01:45:21.318674Z","iopub.status.idle":"2023-04-02T01:45:21.360098Z","shell.execute_reply.started":"2023-04-02T01:45:21.318636Z","shell.execute_reply":"2023-04-02T01:45:21.358646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim\ncriterion = nn.BCELoss()\noptimizer = optim.SGD(model.parameters(), lr=0.001, momentum=0.9)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:23:03.081395Z","iopub.execute_input":"2023-04-02T04:23:03.081877Z","iopub.status.idle":"2023-04-02T04:23:03.089564Z","shell.execute_reply.started":"2023-04-02T04:23:03.081832Z","shell.execute_reply":"2023-04-02T04:23:03.088201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df['cancer']==1]\nimg_data = dataLoader.getimginfo(87)\n\nif img_data['target'] == 1:\n    print(\"Asd\")","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:20:46.972522Z","iopub.execute_input":"2023-04-02T04:20:46.973037Z","iopub.status.idle":"2023-04-02T04:20:46.992997Z","shell.execute_reply.started":"2023-04-02T04:20:46.97299Z","shell.execute_reply":"2023-04-02T04:20:46.99178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResNet(img_channels=256, num_layers=18, block=BasicBlock, num_classes=1)\n\n\nlosses = []\nfor epoch in range (4):\n    i=0\n    running_loss = 0\n    for idx, row in X_train.iterrows():\n        img_data = dataLoader.getimginfo(idx)\n        optimizer.zero_grad()\n        \n        tensor_img = torch.Tensor(img_data['image'])\n        \n        outputs = model(tensor_img.unsqueeze(0))\n        if outputs < 0:\n            print(outputs)\n            outputs[0] = torch.Tensor([0])\n          \n        try:\n            loss = criterion(outputs, img_data['target'].flatten().unsqueeze(1))\n        except RuntimeError as e:\n            print(\"Had an error with outputs\", outputs, '\\n And target:',img_data['target'].flatten().unsqueeze(1))\n            continue\n        \n#         if img_data['target'] == 1:\n#             print(outputs)\n            \n        i+=1\n        running_loss += loss.item()\n        \n        if i%2000 == 1999:\n            print(f'[{epoch + 1}, {i + 1:5d}] loss: {running_loss / 2000:.3f}')\n            losses.append(running_loss / 2000)\n            running_loss = 0.0\n\nprint('Finished Training')\n        #  print(idx)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T06:24:14.286886Z","iopub.execute_input":"2023-04-02T06:24:14.287371Z","iopub.status.idle":"2023-04-02T07:07:10.126204Z","shell.execute_reply.started":"2023-04-02T06:24:14.28733Z","shell.execute_reply":"2023-04-02T07:07:10.124605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = y_test.to_frame()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:05:32.690973Z","iopub.execute_input":"2023-04-02T04:05:32.691406Z","iopub.status.idle":"2023-04-02T04:05:32.697181Z","shell.execute_reply.started":"2023-04-02T04:05:32.691368Z","shell.execute_reply":"2023-04-02T04:05:32.69619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(t[t['cancer'] == 1])/len(t)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:06:18.124442Z","iopub.execute_input":"2023-04-02T04:06:18.124914Z","iopub.status.idle":"2023-04-02T04:06:18.135054Z","shell.execute_reply.started":"2023-04-02T04:06:18.124873Z","shell.execute_reply":"2023-04-02T04:06:18.133761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_saved = model","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:07:29.64234Z","iopub.execute_input":"2023-04-02T07:07:29.642783Z","iopub.status.idle":"2023-04-02T07:07:29.648834Z","shell.execute_reply.started":"2023-04-02T07:07:29.642746Z","shell.execute_reply":"2023-04-02T07:07:29.647539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = 'resnet18_10xcancer_undersample_relu.pth'\ntorch.save(model.state_dict(), path)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:07:30.222436Z","iopub.execute_input":"2023-04-02T07:07:30.222928Z","iopub.status.idle":"2023-04-02T07:07:30.307737Z","shell.execute_reply.started":"2023-04-02T07:07:30.222883Z","shell.execute_reply":"2023-04-02T07:07:30.306164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = 'resnet18_10xcancer_loss_undersample_relu.pth'\ntorch.save(losses, path)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:07:31.252251Z","iopub.execute_input":"2023-04-02T07:07:31.252721Z","iopub.status.idle":"2023-04-02T07:07:31.259219Z","shell.execute_reply.started":"2023-04-02T07:07:31.25268Z","shell.execute_reply":"2023-04-02T07:07:31.257385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = y_test.to_frame(name='iscancer')","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:10:52.951771Z","iopub.execute_input":"2023-04-01T23:10:52.952232Z","iopub.status.idle":"2023-04-01T23:10:52.97619Z","shell.execute_reply.started":"2023-04-01T23:10:52.952192Z","shell.execute_reply":"2023-04-01T23:10:52.974287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancerous_y = y_test[y_test['cancer']==1]\ncancerous_y.iloc[0].values[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:18:51.454281Z","iopub.execute_input":"2023-04-01T23:18:51.454727Z","iopub.status.idle":"2023-04-01T23:18:51.464464Z","shell.execute_reply.started":"2023-04-01T23:18:51.454687Z","shell.execute_reply":"2023-04-01T23:18:51.462919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancerous_x = X_test.loc[cancerous_y.index]","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:13:30.454216Z","iopub.execute_input":"2023-04-01T23:13:30.454654Z","iopub.status.idle":"2023-04-01T23:13:30.46174Z","shell.execute_reply.started":"2023-04-01T23:13:30.454614Z","shell.execute_reply":"2023-04-01T23:13:30.460347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(cancerous_x), len(cancerous_y))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:13:49.551815Z","iopub.execute_input":"2023-04-01T23:13:49.55227Z","iopub.status.idle":"2023-04-01T23:13:49.558861Z","shell.execute_reply.started":"2023-04-01T23:13:49.552227Z","shell.execute_reply":"2023-04-01T23:13:49.557467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_sub = X_test.head(200)\ny_test_sub = y_test.head(200)\n\nfinal_X = cancerous_x.append(x_test_sub)\nfinal_y = cancerous_y.append(y_test_sub)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:27:15.309498Z","iopub.execute_input":"2023-04-01T23:27:15.309954Z","iopub.status.idle":"2023-04-01T23:27:15.321341Z","shell.execute_reply.started":"2023-04-01T23:27:15.309912Z","shell.execute_reply":"2023-04-01T23:27:15.32007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_X","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:28:15.707302Z","iopub.execute_input":"2023-04-01T23:28:15.708098Z","iopub.status.idle":"2023-04-01T23:28:15.735259Z","shell.execute_reply.started":"2023-04-01T23:28:15.70805Z","shell.execute_reply":"2023-04-01T23:28:15.734311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(final_y[final_y['cancer'] == 1]), len(final_y[final_y['cancer'] == 0]), len(final_y))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:29:05.732328Z","iopub.execute_input":"2023-04-01T23:29:05.732757Z","iopub.status.idle":"2023-04-01T23:29:05.740151Z","shell.execute_reply.started":"2023-04-01T23:29:05.732719Z","shell.execute_reply":"2023-04-01T23:29:05.739135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(y_test==1), len(y_test == 0))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T04:02:53.009552Z","iopub.execute_input":"2023-04-02T04:02:53.010071Z","iopub.status.idle":"2023-04-02T04:02:53.01875Z","shell.execute_reply.started":"2023-04-02T04:02:53.010026Z","shell.execute_reply":"2023-04-02T04:02:53.017315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ni=0\ncorrect = 0\nmax_output_val = 0\npred_for_max = None\nfor idx, row in X_test.iterrows():\n    img_data = dataLoader.getimginfo(idx)\n    tensor_img = torch.Tensor(img_data['image'])\n    outputs = model(tensor_img.unsqueeze(0))\n#     print(outputs, y_test.iloc[i])\n    if outputs > max_output_val:\n        max_output_val = outputs\n        pred_for_max = y_test.iloc[i]\n    if outputs > 0.5:\n        outputs = 1\n    else:\n        outputs = 0\n    \n    if outputs == y_test.iloc[i]:\n        correct+=1\n        \n    if i%100 == 0:\n        print(i, \"Completed\")\n    i+=1\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:08:43.445762Z","iopub.execute_input":"2023-04-02T07:08:43.44623Z","iopub.status.idle":"2023-04-02T07:11:25.450431Z","shell.execute_reply.started":"2023-04-02T07:08:43.446191Z","shell.execute_reply":"2023-04-02T07:11:25.448387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(len(final_X), correct)\nprint(\"ACCURACY\", correct/len(X_test))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:13:37.297033Z","iopub.execute_input":"2023-04-02T07:13:37.297569Z","iopub.status.idle":"2023-04-02T07:13:37.305522Z","shell.execute_reply.started":"2023-04-02T07:13:37.297442Z","shell.execute_reply":"2023-04-02T07:13:37.303811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(max_output_val, pred_for_max)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:14:06.145127Z","iopub.execute_input":"2023-04-02T07:14:06.146252Z","iopub.status.idle":"2023-04-02T07:14:06.156213Z","shell.execute_reply.started":"2023-04-02T07:14:06.14617Z","shell.execute_reply":"2023-04-02T07:14:06.154568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(cancerous_x), correct)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:21:07.632285Z","iopub.execute_input":"2023-04-01T23:21:07.633861Z","iopub.status.idle":"2023-04-01T23:21:07.641637Z","shell.execute_reply.started":"2023-04-01T23:21:07.633806Z","shell.execute_reply":"2023-04-01T23:21:07.64024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"ACCURACY\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\ncorrect = 0\nfor idx, row in X_test.iterrows():\n    img_data = dataLoader.getimginfo(idx)\n    tensor_img = torch.Tensor(img_data['image'])\n    outputs = model(tensor_img.unsqueeze(0))\n    \n    if outputs > 0.5:\n        outputs = 1\n    else:\n        outputs = 0\n        \n    if outputs == y_test.iloc[i]:\n        correct+=1\n        \n    if i%500 == 0:\n        print(i, \"Completed\")\n    i+=1\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:01:03.218906Z","iopub.execute_input":"2023-04-01T23:01:03.219993Z","iopub.status.idle":"2023-04-01T23:05:17.566784Z","shell.execute_reply.started":"2023-04-01T23:01:03.219949Z","shell.execute_reply":"2023-04-01T23:05:17.565751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:07:59.65665Z","iopub.execute_input":"2023-04-01T23:07:59.657235Z","iopub.status.idle":"2023-04-01T23:07:59.665032Z","shell.execute_reply.started":"2023-04-01T23:07:59.657164Z","shell.execute_reply":"2023-04-01T23:07:59.663765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correct","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:07:54.37327Z","iopub.execute_input":"2023-04-01T23:07:54.373725Z","iopub.status.idle":"2023-04-01T23:07:54.381892Z","shell.execute_reply.started":"2023-04-01T23:07:54.373683Z","shell.execute_reply":"2023-04-01T23:07:54.380519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(correct / len(X_test))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T23:08:15.964246Z","iopub.execute_input":"2023-04-01T23:08:15.964689Z","iopub.status.idle":"2023-04-01T23:08:15.970354Z","shell.execute_reply.started":"2023-04-01T23:08:15.96465Z","shell.execute_reply":"2023-04-01T23:08:15.969033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ROC CURVE","metadata":{}},{"cell_type":"code","source":"# thresholds = {\n#     0.005: {},\n#     0.050: {},\n#     0.10: { },\n#     0.002: {},\n#     0.0024: {\n\n#    }, \n# }\n# keep track of every test prediction and its actual target value\n# list of tuples\n# (prediction value, target)\ndef get_predictions():\n    test_predictions = []\n    i=0\n    for idx, row in X_test.iterrows():\n        img_data = dataLoader.getimginfo(idx)\n        tensor_img = torch.Tensor(img_data['image'])\n        outputs = model(tensor_img.unsqueeze(0))\n        test_predictions.append((outputs.item(), y_test.iloc[i]))\n        i+=1\n        \n    return test_predictions\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:31:57.370187Z","iopub.execute_input":"2023-04-02T07:31:57.370682Z","iopub.status.idle":"2023-04-02T07:31:57.37942Z","shell.execute_reply.started":"2023-04-02T07:31:57.370638Z","shell.execute_reply":"2023-04-02T07:31:57.377896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = get_predictions()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:31:58.443573Z","iopub.execute_input":"2023-04-02T07:31:58.444072Z","iopub.status.idle":"2023-04-02T07:34:36.008638Z","shell.execute_reply.started":"2023-04-02T07:31:58.444023Z","shell.execute_reply":"2023-04-02T07:34:36.007041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(predictions)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T07:34:36.011267Z","iopub.execute_input":"2023-04-02T07:34:36.013249Z","iopub.status.idle":"2023-04-02T07:34:36.033973Z","shell.execute_reply.started":"2023-04-02T07:34:36.013185Z","shell.execute_reply":"2023-04-02T07:34:36.032138Z"},"trusted":true},"execution_count":null,"outputs":[]}]}