{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6420245,"sourceType":"datasetVersion","datasetId":3694516},{"sourceId":6454768,"sourceType":"datasetVersion","datasetId":3726958}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nfrom collections import defaultdict\nfrom pathlib import Path\nfrom torch.utils.data import Dataset,DataLoader,random_split\nfrom torchvision.models import resnet50,ResNet50_Weights,efficientnet_v2_m,EfficientNet_V2_M_Weights,densenet161,DenseNet161_Weights\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom tqdm import tqdm\nfrom tabulate import tabulate\nfrom sklearn.metrics import accuracy_score, roc_auc_score\n\n\n\nimport cv2\nimport numpy as np # linear algebra\nimport os\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pydicom\nimport torch\nimport torchvision.transforms as transforms\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         if filename[-3:] == 'csv':\n#             print(os.path.join(dirname, filename))\n#         else:\n#             break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-31T12:12:52.153589Z","iopub.execute_input":"2024-10-31T12:12:52.153862Z","iopub.status.idle":"2024-10-31T12:12:57.515152Z","shell.execute_reply.started":"2024-10-31T12:12:52.153837Z","shell.execute_reply":"2024-10-31T12:12:57.514157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv_file = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv'\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.517134Z","iopub.execute_input":"2024-10-31T12:12:57.517873Z","iopub.status.idle":"2024-10-31T12:12:57.555332Z","shell.execute_reply.started":"2024-10-31T12:12:57.517845Z","shell.execute_reply":"2024-10-31T12:12:57.554455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_folder = '/kaggle/input/rsna-abdotrauma-rgb-10mmthick/reduced_256_thickness_10'\ndataset_folder","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.556392Z","iopub.execute_input":"2024-10-31T12:12:57.556655Z","iopub.status.idle":"2024-10-31T12:12:57.562435Z","shell.execute_reply.started":"2024-10-31T12:12:57.556632Z","shell.execute_reply":"2024-10-31T12:12:57.561461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preparation** ##","metadata":{}},{"cell_type":"code","source":"def get_patient_and_labels(csv,train_images):\n    \n    folders = [f for f in os.listdir(train_images) if os.path.isdir(os.path.join(train_images, f))]\n\n    patients = defaultdict(list)\n    for f in folders:\n        instances = Path(train_images + '/' + f)\n        for files in instances.iterdir():\n            for file in files.iterdir():\n                patients[f].append(file)\n    \n    df= pd.read_csv(csv)\n    patient_ids = list(df['patient_id'])\n    #all_labels = np.array(df.drop(['patient_id'], axis=1))\n                                #bowel  #extr  #kid  #liver   #spleen\n    #all_labels = list(map(lambda x:(x[0:2],x[2:4],x[4:7],x[7:10],x[10:13]),all_labels))\n    all_labels = list(map(lambda x: df[df.patient_id == int(x)].values[0][1:-1],patient_ids))\n\n    #combine patient_ids and labels\n    combined = dict(zip(patient_ids,all_labels))\n\n    image_paths = []\n    labels = []\n\n    for key,value in combined.items(): \n        if str(key) in patients:   #check if the image is already processed\n            img_paths = patients.get(str(key))\n            for path in img_paths:\n                image_paths.append(path)\n                labels.append(value)\n\n\n    return image_paths, labels\n\n\n\nclass RSNADataset(Dataset):\n    def __init__(self,csv_file,train_images,transform=False):\n        self.patient_images,self.labels = get_patient_and_labels(csv_file,train_images)\n        \n        if transform == True:\n            self.transform = transforms.Compose([\n                transforms.ToPILImage(),\n                transforms.Resize((400, 400)),\n                transforms.RandomHorizontalFlip(p=0.5),\n                transforms.RandomRotation(degrees=45),\n                transforms.ToTensor(),\n            ])\n    \n        \n    def __len__(self):\n        return len(self.patient_images)\n    \n    def __getitem__(self, index):\n        image_path = self.patient_images[index]\n        image = cv2.imread(str(image_path))\n        # convert the image from BGR to RGB color format\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        # apply image transforms\n        #image = self.transform(image)\n        image = torch.tensor(image, dtype=torch.float32)\n        # change the image shape to c,h,w from h,w,c\n        image = image.permute(2,0,1)\n        #print(f\"shape of image: {image.shape}\")\n        labels = self.labels[index]\n        \n        \n        #         bowel = np.argmax(labels[0:2], keepdims = True)\n#         extravasation = np.argmax(labels[2:4], keepdims = True)\n#         kidney = np.argmax(labels[4:7], keepdims = False)\n#         liver = np.argmax(labels[7:10], keepdims = False)\n#         spleen = np.argmax(labels[10:], keepdims = False)\n        \n        \n        \n        \n        bowel = torch.tensor(labels[0:2], dtype=torch.float32)\n        extravasation = torch.tensor(labels[2:4], dtype=torch.float32)\n        kidney = torch.tensor(labels[4:7], dtype=torch.float32)\n        liver = torch.tensor(labels[7:10], dtype=torch.float32)\n        spleen = torch.tensor(labels[10:], dtype=torch.float32)\n\n       \n        \n\n        return {\n            'image': image,\n            'bowel': bowel,\n            'extravasation': extravasation,\n            'kidney': kidney,\n            'liver' : liver,\n            'spleen' : spleen\n        }\n","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.565038Z","iopub.execute_input":"2024-10-31T12:12:57.565315Z","iopub.status.idle":"2024-10-31T12:12:57.581597Z","shell.execute_reply.started":"2024-10-31T12:12:57.565292Z","shell.execute_reply":"2024-10-31T12:12:57.580741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model** ##","metadata":{}},{"cell_type":"code","source":"class MultiHeadResNet50(nn.Module):\n    def __init__(self):\n        super().__init__()\n        resnet = resnet50(weights=ResNet50_Weights.IMAGENET1K_V2)\n        #efficient_net = efficientnet_v2_m(weights = EfficientNet_V2_M_Weights.IMAGENET1K_V1 )\n        #densenet = densenet161(weights = DenseNet161_Weights.IMAGENET1K_V1)\n        # Remove the fully connected layers (the classifier)\n        # to keep only the feature extraction part\n        modules = list(resnet.children())[:-2]\n        #modules = list(densenet.children())[:-1]\n        self.feature_extractor = nn.Sequential(*modules)\n        #maxpool_layer = nn.MaxPool2d(kernel_size=2, stride=2)\n        \n        num_features = resnet.fc.in_features\n        #num_features = 2208\n        # change the final layers according to the number of categories\n        self.bowel = nn.Linear(num_features, 2) \n        self.extravasation = nn.Linear(num_features, 2) \n        self.kidney = nn.Linear(num_features, 3) \n        self.liver = nn.Linear(num_features, 3) \n        self.spleen = nn.Linear(num_features, 3) \n\n\n    def forward(self, x):\n          # get the batch size only, ignore (c, h, w)\n        batch, _, _, _ = x.shape\n        #feature extraction\n        x = self.feature_extractor(x)\n        features = F.adaptive_avg_pool2d(x, 1).reshape(batch, -1)\n        #output logits\n        bowel = self.bowel(features)\n        #print(f\"bowel shape is: {bowel.shape}\")\n        extravasation = self.extravasation(features)\n        kidney = self.kidney(features)\n        liver = self.liver(features)\n        spleen = self.spleen(features)\n        \n        return bowel,extravasation, kidney, liver, spleen","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.582959Z","iopub.execute_input":"2024-10-31T12:12:57.583282Z","iopub.status.idle":"2024-10-31T12:12:57.595912Z","shell.execute_reply.started":"2024-10-31T12:12:57.583252Z","shell.execute_reply":"2024-10-31T12:12:57.595041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Training Preparation** ##","metadata":{}},{"cell_type":"code","source":"# custom loss function for multi-head multi-category classification\ndef loss_fn(outputs, targets,device):\n    b, e, k, l, s = outputs\n    bowel, extravasation, kidney, liver, spleen = targets\n\n    #bce_b = nn.BCEWithLogitsLoss(pos_weight = torch.tensor([2.0]).to(device))\n    #bce_e = nn.BCEWithLogitsLoss(pos_weight = torch.tensor([4.0]).to(device))\n    \n    bce_b = nn.CrossEntropyLoss(label_smoothing = 0.05, weight = torch.tensor([1.0, 2.0]).to(device))\n    bce_e = nn.CrossEntropyLoss(label_smoothing = 0.05, weight = torch.tensor([1.0, 2.0]).to(device))\n    cce = nn.CrossEntropyLoss(label_smoothing = 0.05, weight = torch.tensor([1.0, 2.0, 4.0]).to(device))\n    \n    b_loss = bce_b(b, bowel.float())\n    e_loss = bce_e(e, extravasation.float())\n    k_loss = cce(k, kidney)\n    l_loss = cce(l, liver)\n    s_loss = cce(s, spleen)\n    \n    total_loss = b_loss + e_loss + l_loss + k_loss + s_loss\n    #average_loss = total_loss / 5\n    \n    return total_loss\n\ndef spliting_data(split_ratio_test,dataset):\n    \n    test_percent = int(split_ratio_test * len(dataset))\n    train_size = len(dataset) - test_percent\n    valid_size = int(len(dataset) - train_size)\n    train_set,valid_set = random_split(dataset,[train_size,valid_size])\n    \n    return train_set,valid_set\n\n\n# training function\ndef training_loop(model,dataloader, device, optimizer, loss_fn, dataset, metrics):\n    model.train()\n    counter = 0\n    train_running_loss = 0.0\n    for i, data in tqdm(enumerate(dataloader), total=int(len(dataset)/dataloader.batch_size)):\n        counter += 1\n        \n        # extract the features and labels\n        image = data['image'].to(device)\n        bowel = data['bowel'].to(device)\n        extravasation = data['extravasation'].to(device)\n        kidney = data['kidney'].to(device)\n        liver = data['liver'].to(device)\n        spleen = data['spleen'].to(device)\n\n\n        \n        # zero-out the optimizer gradients\n        optimizer.zero_grad()\n        \n        b,e,k,l,s = model(image)\n        outputs= (b,e,k,l,s)\n        targets = (bowel, extravasation, kidney, liver, spleen)\n        loss = loss_fn(outputs, targets,device)\n\n        train_running_loss += loss.item()\n        \n        # backpropagation\n        loss.backward()\n        # update optimizer parameters\n        optimizer.step()\n        \n        #break\n      \n          \n    train_loss = train_running_loss / counter\n    \n\n    return train_loss\n\n\n# validation function\ndef validating_loop(model, dataloader, device, loss_fn, dataset, metrics):\n    model.eval()\n    counter = 0\n    val_running_loss = 0.0\n    for i, data in tqdm(enumerate(dataloader), total=int(len(dataset)/dataloader.batch_size)):\n        counter += 1\n        \n        # extract the features and labels\n\n        image = data['image'].to(device)\n        bowel = data['bowel'].to(device)\n        extravasation = data['extravasation'].to(device)\n        kidney = data['kidney'].to(device)\n        liver = data['liver'].to(device)\n        spleen = data['spleen'].to(device)\n\n        \n        b,e,k,l,s = model(image)\n        outputs= (b,e,k,l,s)\n        targets = (bowel, extravasation, kidney, liver, spleen)\n        loss = loss_fn(outputs, targets,device)\n        val_running_loss += loss.item()\n\n\n        #calculate training metrics\n        metrics['bowel'].update(b, bowel)\n        metrics['extravasation'].update(e, extravasation)\n        metrics['kidney'].update(k, kidney)\n        metrics['liver'].update(l, liver)\n        metrics['spleen'].update(s, spleen)\n        \n        #break\n        \n    val_loss = val_running_loss / counter\n    return val_loss\n","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.597028Z","iopub.execute_input":"2024-10-31T12:12:57.597431Z","iopub.status.idle":"2024-10-31T12:12:57.616269Z","shell.execute_reply.started":"2024-10-31T12:12:57.597407Z","shell.execute_reply":"2024-10-31T12:12:57.615424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Evaluation Metrics** ##","metadata":{}},{"cell_type":"code","source":"class MetricsCalculator:\n    \n    def __init__(self, mode = 'binary'):\n        \n        self.probabilities = []\n        self.predictions = []\n        self.targets = []\n        \n        self.mode = mode\n\n    def update(self, logits, target):\n        \"\"\"\n        Update the metrics calculator with predicted values and corresponding targets.\n        \n        Args:\n            predicted (torch.Tensor): Predicted values.\n            target (torch.Tensor): Ground truth targets.\n        \"\"\"\n        if self.mode == 'binary':\n            probabilities = torch.sigmoid(logits)\n            predicted = (probabilities > 0.5)\n        else:\n            probabilities = F.softmax(logits, dim = 1)\n            predicted = torch.argmax(probabilities, dim=1)\n            target_softmaxed = torch.argmax(target, dim=1)\n            \n        self.probabilities.extend(probabilities.detach().cpu().numpy())\n        self.predictions.extend(predicted.detach().cpu().numpy())\n        self.targets.extend(target_softmaxed.detach().cpu().numpy())\n\n    \n\n    def reset(self):\n        \"\"\"Reset the stored predictions and targets.\"\"\"\n        \n        self.probabilities = []\n        self.predictions = []\n        self.targets = []\n\n\n    def compute_accuracy(self):\n        \"\"\"\n        Compute the accuracy metric.\n        \n        Returns:\n            float: Accuracy.\n        \"\"\"\n        return accuracy_score(self.targets, self.predictions)\n    \n\n    def compute_auc(self):\n        \"\"\"\n        Compute the AUC (Area Under the Curve) metric.\n        \n        Returns:\n            float: AUC.\n        \"\"\"\n        if self.mode == 'multi':\n            return roc_auc_score(self.targets, self.probabilities, multi_class = 'ovo', labels=[0, 1, 2])\n    \n        else:\n            return roc_auc_score(self.targets, self.probabilities)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.617422Z","iopub.execute_input":"2024-10-31T12:12:57.6177Z","iopub.status.idle":"2024-10-31T12:12:57.633312Z","shell.execute_reply.started":"2024-10-31T12:12:57.617677Z","shell.execute_reply":"2024-10-31T12:12:57.632387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Training** ##","metadata":{}},{"cell_type":"code","source":" # define the computation device\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n # initialize the model\nmodel = MultiHeadResNet50().to(device)\n # learning parameters\nlr = 0.001\n\noptimizer = optim.SGD(params=model.parameters(), lr=lr)\n# optimizer = optim.Adadelta(params=model.parameters(), lr=lr)\n#optimizer = optim.Adam(params=model.parameters(), lr=lr)\n# optimizer = optim.Adagrad(params=model.parameters(), lr=lr)\n#optimizer = optim.RMSprop(params=model.parameters(), lr=lr)\n\nscheduler = ReduceLROnPlateau(optimizer, mode='min', patience=5, factor=0.5, verbose=True)\n\ncriterion = loss_fn\nbatch_size = 16\nepochs = 1 #20\nsplit_ratio_test = 0.2\n\ndataset = RSNADataset(csv_file=csv_file,train_images = dataset_folder)\ntrain_set,valid_set = spliting_data(split_ratio_test,dataset)\n\ntrain_dl = DataLoader(train_set, batch_size, shuffle=True)\nvalid_dl = DataLoader(valid_set, batch_size )\n\n\n# initialize metrics objects\nmetrics = {\n    \"bowel\":MetricsCalculator('multi'),\n    \"extravasation\": MetricsCalculator('multi'),\n    \"kidney\" : MetricsCalculator('multi'),\n    \"liver\" : MetricsCalculator('multi'),\n    \"spleen\" : MetricsCalculator('multi')\n        }\n\n# start the training\nprev_val_best_loss = float('inf')\n\n#train_loss, val_loss = [], []\nfor epoch in range(epochs):\n    print(f\"Epoch {epoch+1} of {epochs}\")\n    print(\"training\")\n    train_epoch_loss = training_loop(model,train_dl, device, optimizer, loss_fn, dataset, metrics)\n        \n    print(\"validating\")\n    val_epoch_loss = validating_loop(model,valid_dl, device, loss_fn, dataset, metrics)\n\n    scheduler.step(val_epoch_loss)\n        \n    if val_epoch_loss < prev_val_best_loss:\n        prev_val_best_loss = val_epoch_loss\n        print(\"Validation Loss improved, Saving Model...\")\n        torch.save(model, f'resnet_50_{val_epoch_loss:.3f}.pth')\n\n\n     # accuracy and auc data\n    metrics_data = [\n        [\"Bowel\", \n         metrics['bowel'].compute_accuracy()],\n        [\"Extravasation\",\n         metrics['extravasation'].compute_accuracy()],\n        [\"Kidney\",\n         metrics['kidney'].compute_accuracy()],\n        [\"Liver\",\n         metrics['liver'].compute_accuracy()],\n        [\"Spleen\",\n         metrics['spleen'].compute_accuracy()]\n    ]\n    \n    \n    print('evaluation')\n    print(tabulate(metrics_data, headers=[\"\", \"Validation Acc\"]))\n    print(f\"Train Loss (Mean): {train_epoch_loss:.4f}\")\n    print(f\"Validation Loss (Mean): {val_epoch_loss:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-31T12:12:57.636524Z","iopub.execute_input":"2024-10-31T12:12:57.636785Z","iopub.status.idle":"2024-10-31T13:08:27.713203Z","shell.execute_reply.started":"2024-10-31T12:12:57.63676Z","shell.execute_reply":"2024-10-31T13:08:27.712144Z"},"trusted":true},"execution_count":null,"outputs":[]}]}