{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6331238,"sourceType":"datasetVersion","datasetId":3644255}],"dockerImageVersionId":30528,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import packages\nimport os\nimport pickle\nfrom tqdm.notebook import tqdm\nimport random\nfrom tabulate import tabulate\n\nimport cv2\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\nimport torchvision.transforms.v2 as t\n\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom torch.optim import Adam\nfrom torchvision import models\nfrom torchvision.transforms.v2 import Resize, Compose, RandomHorizontalFlip, ColorJitter, RandomAffine, RandomErasing, ToTensor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-15T18:41:18.526763Z","iopub.execute_input":"2024-11-15T18:41:18.527476Z","iopub.status.idle":"2024-11-15T18:41:18.535582Z","shell.execute_reply.started":"2024-11-15T18:41:18.527444Z","shell.execute_reply":"2024-11-15T18:41:18.534441Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_IMG_PATH = '/kaggle/input/rsna-2023-atd-reduced-256-5mm/reduced_256_tickness_5'\nTEST_IMG_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images'\nTRAIN_DF_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv'","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.537269Z","iopub.execute_input":"2024-11-15T18:41:18.537558Z","iopub.status.idle":"2024-11-15T18:41:18.55333Z","shell.execute_reply.started":"2024-11-15T18:41:18.537533Z","shell.execute_reply":"2024-11-15T18:41:18.55245Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fetch_img_paths(train_img_path):\n    img_paths = []\n    \n    print('Scanning directories...')\n    for patient in tqdm(os.listdir(train_img_path)):\n        for scan in os.listdir(os.path.join(TRAIN_IMG_PATH, patient)):\n            scans = []\n            for img in os.listdir(os.path.join(TRAIN_IMG_PATH, patient, scan)):\n                scans.append(os.path.join(TRAIN_IMG_PATH, patient, scan, img))\n            \n            img_paths.append(scans)\n            \n    return img_paths","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.554448Z","iopub.execute_input":"2024-11-15T18:41:18.554712Z","iopub.status.idle":"2024-11-15T18:41:18.566903Z","shell.execute_reply.started":"2024-11-15T18:41:18.554672Z","shell.execute_reply":"2024-11-15T18:41:18.566138Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_elements_with_spacing(input_list, spacing):\n    \n    \"\"\"\n    Selects elements with a specified spacing from a given list.\n\n    Args:\n        input_list (list): The input list from which elements will be selected.\n        spacing (int): The spacing between selected elements.\n\n    Returns:\n        list: A list of selected elements from the input list.\n\n    Raises:\n        ValueError: If the input list does not contain at least 4 * spacing elements.\n    \"\"\"\n \n    if len(input_list) < spacing * 4:\n        raise ValueError(\"List should contain at least 4 * spacing elements.\")\n        \n        \n    # We want to select elements in the middle part of the abdomen\n    lower_bound = int(len(input_list) * 0.4)\n    upper_bound = int(len(input_list) * 0.6)\n\n    spacing = (upper_bound - lower_bound) // 3\n    \n    # start_index = random.randint(lower_bound, upper_bound)\n    \n    selected_indices = [lower_bound, lower_bound + spacing, lower_bound + (2*spacing), upper_bound]\n    \n    selected_elements = [input_list[index] for index in selected_indices]\n    \n    return selected_elements","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.568725Z","iopub.execute_input":"2024-11-15T18:41:18.569006Z","iopub.status.idle":"2024-11-15T18:41:18.579562Z","shell.execute_reply.started":"2024-11-15T18:41:18.568982Z","shell.execute_reply":"2024-11-15T18:41:18.57877Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def standardize_pixel_array(dicom_image):\n    \"\"\"\n    Standardizes a DICOM pixel array by applying various transformations.\n    \n    Args:\n        dicom_path (str): Path to the DICOM image file.\n        \n    Returns:\n        np.ndarray: The standardized pixel array of the DICOM image.\n    \"\"\"\n    pixel_array = dicom_image.pixel_array\n    \n    if dicom_image.PixelRepresentation == 1:\n        bit_shift = dicom_image.BitsAllocated - dicom_image.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dicom_image)\n\n    if dicom_image.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n\n    # transform to hounsfield units\n    intercept = dicom_image.RescaleIntercept\n    slope = dicom_image.RescaleSlope\n    pixel_array = pixel_array * slope + intercept\n\n    # windowing\n    window_center = int(dicom_image.WindowCenter)\n    window_width = int(dicom_image.WindowWidth)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    pixel_array = pixel_array.copy()\n    pixel_array[pixel_array < img_min] = img_min\n    pixel_array[pixel_array > img_max] = img_max\n\n    # normalization\n    if pixel_array.max() == pixel_array.min():\n        pixel_array = np.zeros_like(pixel_array)  # Handle case of constant array\n    else:\n        pixel_array = (pixel_array - pixel_array.min()) / (pixel_array.max() - pixel_array.min())\n\n    return pixel_array","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.5806Z","iopub.execute_input":"2024-11-15T18:41:18.580846Z","iopub.status.idle":"2024-11-15T18:41:18.601939Z","shell.execute_reply.started":"2024-11-15T18:41:18.580823Z","shell.execute_reply":"2024-11-15T18:41:18.60104Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_jpeg(jpeg_path):\n    \n    img = cv2.imread(jpeg_path)\n    greyscale = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)/255\n    \n    return greyscale","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.603114Z","iopub.execute_input":"2024-11-15T18:41:18.60339Z","iopub.status.idle":"2024-11-15T18:41:18.617836Z","shell.execute_reply.started":"2024-11-15T18:41:18.603344Z","shell.execute_reply":"2024-11-15T18:41:18.617019Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dataset\nclass AbdominalData(Dataset):\n    \"\"\"\n    Custom dataset class for handling abdominal trauma data classification.\n    \n    Args:\n        df_path (str): Path to the CSV file containing patient labels.\n        current_fold (int): The current fold for cross-validation.\n        num_fold (int, optional): Total number of folds for cross-validation. Default is 5.\n    \"\"\"\n    \n    def __init__(self, df_path, current_fold, num_fold = 5):\n        \n        super().__init__()\n        \n        # collect all the image instance paths\n        self.img_paths = fetch_img_paths(TRAIN_IMG_PATH)\n                \n        self.df = pd.read_csv(df_path)\n        \n        self.num_fold = num_fold\n        self.current_fold = current_fold\n        self.kf = KFold(n_splits=num_fold)\n        \n        self.transform = Compose([\n#                             Resize((256, 256), antialias=True),\n                            RandomHorizontalFlip(),  # Randomly flip images left-right\n                            ColorJitter(brightness=0.2),  # Randomly adjust brightness\n                            ColorJitter(contrast=0.2),  # Randomly adjust contrast\n                            RandomAffine(degrees=0, shear=10),  # Apply shear transformation\n                            RandomAffine(degrees=0, scale=(0.8, 1.2)),  # Apply zoom transformation\n                            RandomErasing(p=0.2, scale=(0.02, 0.2)), # Coarse dropout\n                            ToTensor(),\n                        ])\n    \n    def __len__(self):\n        \"\"\"\n        Returns the total number of samples in the dataset.\n        \"\"\"\n        \n        return len(self.img_paths)\n    \n    def __getitem__(self, idx):\n        \"\"\"\n        Retrieves a sample from the dataset by index.\n        \n        Args:\n            idx (int): Index of the dataset to retrieve.\n        \n        Returns:\n            dict: A dictionary containing the image data and labels for different abdominal structures.\n        \"\"\"\n        \n        # sample 4 image instances\n        dicom_images = select_elements_with_spacing(self.img_paths[idx],\n                                                    spacing = 2)\n        patient_id = dicom_images[0].split('/')[-3]\n        images = []\n        \n        for d in dicom_images:\n            image = preprocess_jpeg(d)\n            images.append(image)\n            \n        images = np.stack(images)\n        image = torch.tensor(images, dtype = torch.float).unsqueeze(dim = 1)\n        \n        image = self.transform(image).squeeze(dim = 1)\n        \n        label = self.df[self.df.patient_id == int(patient_id)].values[0][1:-1]\n        \n        # labels\n        bowel = np.argmax(label[0:2], keepdims = True)\n        extravasation = np.argmax(label[2:4], keepdims = True)\n        kidney = np.argmax(label[4:7], keepdims = False)\n        liver = np.argmax(label[7:10], keepdims = False)\n        spleen = np.argmax(label[10:], keepdims = False)\n        \n        \n        return {\n            'image': image,\n            'bowel': bowel,\n            'extravasation': extravasation,\n            'kidney': kidney,\n            'liver': liver,\n            'spleen': spleen,\n        }\n    \n    def get_splits(self):\n        \"\"\"\n        Splits the dataset into training and validation subsets based on the current fold.\n        \n        Returns:\n            tuple: A tuple containing the training and validation subsets.\n        \"\"\"\n        \n        fold_data = list(self.kf.split(self.img_paths))\n        train_indices, val_indices = fold_data[self.current_fold]\n\n        train_data = self._get_subset(train_indices)\n        val_data = self._get_subset(val_indices)\n        \n        return train_data, val_data\n\n    def _get_subset(self, indices):\n        \"\"\"\n        Returns a subset of the dataset based on the provided indices.\n        \n        Args:\n            indices (list): List of indices to include in the subset.\n        \n        Returns:\n            Subset: A subset of the dataset.\n        \"\"\"\n        return Subset(self, indices)","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.699781Z","iopub.execute_input":"2024-11-15T18:41:18.700103Z","iopub.status.idle":"2024-11-15T18:41:18.717893Z","shell.execute_reply.started":"2024-11-15T18:41:18.700076Z","shell.execute_reply":"2024-11-15T18:41:18.717042Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MetricsCalculator:\n    \n    def __init__(self, mode = 'binary'):\n        \n        self.probabilities = []\n        self.predictions = []\n        self.targets = []\n        \n        self.mode = mode\n    \n    def update(self, logits, target):\n        \"\"\"\n        Update the metrics calculator with predicted values and corresponding targets.\n        \n        Args:\n            predicted (torch.Tensor): Predicted values.\n            target (torch.Tensor): Ground truth targets.\n        \"\"\"\n        if self.mode == 'binary':\n            probabilities = torch.sigmoid(logits)\n            predicted = (probabilities > 0.5)\n        else:\n            probabilities = F.softmax(logits, dim = 1)\n            predicted = torch.argmax(probabilities, dim=1)\n            \n        self.probabilities.extend(probabilities.detach().cpu().numpy())\n        self.predictions.extend(predicted.detach().cpu().numpy())\n        self.targets.extend(target.detach().cpu().numpy())\n    \n    def reset(self):\n        \"\"\"Reset the stored predictions and targets.\"\"\"\n        \n        self.probabilities = []\n        self.predictions = []\n        self.targets = []\n    \n    def compute_accuracy(self):\n        \"\"\"\n        Compute the accuracy metric.\n        \n        Returns:\n            float: Accuracy.\n        \"\"\"\n        return accuracy_score(self.targets, self.predictions)\n    \n    def compute_auc(self):\n        \"\"\"\n        Compute the AUC (Area Under the Curve) metric.\n        \n        Returns:\n            float: AUC.\n        \"\"\"\n        if self.mode == 'multi':\n            return roc_auc_score(self.targets, self.probabilities, multi_class = 'ovo', labels=[0, 1, 2])\n    \n        else:\n            return roc_auc_score(self.targets, self.probabilities)","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.719886Z","iopub.execute_input":"2024-11-15T18:41:18.72023Z","iopub.status.idle":"2024-11-15T18:41:18.734811Z","shell.execute_reply.started":"2024-11-15T18:41:18.720204Z","shell.execute_reply":"2024-11-15T18:41:18.734057Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\n\nclass SelfAttention(nn.Module):\n    def __init__(self, in_channels):\n        super(SelfAttention, self).__init__()\n        self.attention = nn.MultiheadAttention(embed_dim=in_channels, num_heads=1, batch_first=True)\n\n    def forward(self, x):\n        batch_size, channels, height, width = x.size()\n        x = x.view(batch_size, channels, -1).transpose(1, 2)  # (batch_size, height*width, channels)\n        attn_output, _ = self.attention(x, x, x)\n        x = attn_output.transpose(1, 2).view(batch_size, channels, height, width)  # (batch_size, channels, height, width)\n        return x\n\nclass CNNModel(nn.Module):\n    def __init__(self):\n        super(CNNModel, self).__init__()\n        \n        # Chuyển đầu vào có 4 kênh thành 3 kênh\n        self.input = nn.Conv2d(4, 3, kernel_size=3, padding=1)  # Giữ lớp này như bạn yêu cầu\n        \n        # Sử dụng EfficientNetB0 làm backbone\n        model = models.efficientnet_b0(weights='IMAGENET1K_V1')\n        self.features = model.features\n        \n        # Thêm self-attention\n        self.self_attention = SelfAttention(1280)  # Sử dụng self-attention sau khi trích xuất đặc trưng\n        self.avgpool = nn.AdaptiveAvgPool2d((1, 1))\n        \n        # Các lớp phân loại\n        self.bowel = nn.Linear(1280, 1)\n        self.extravasation = nn.Linear(1280, 1)\n        self.kidney = nn.Linear(1280, 3)\n        self.liver = nn.Linear(1280, 3)\n        self.spleen = nn.Linear(1280, 3)\n\n    def forward(self, x):\n        # Chuyển đầu vào có 4 kênh thành 3 kênh\n        x = self.input(x)\n        \n        # Trích xuất đặc trưng từ EfficientNetB0\n        x = self.features(x)\n        \n        # Áp dụng self-attention\n        x = self.self_attention(x)\n        \n        # Adaptive average pooling\n        x = self.avgpool(x)\n        \n        # Chuyển đổi thành vector một chiều\n        x = torch.flatten(x, 1)\n        \n        # Phân loại các nhãn\n        bowel = self.bowel(x)\n        extravasation = self.extravasation(x)\n        kidney = self.kidney(x)\n        liver = self.liver(x)\n        spleen = self.spleen(x)\n        \n        return bowel, extravasation, kidney, liver, spleen\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.736125Z","iopub.execute_input":"2024-11-15T18:41:18.73641Z","iopub.status.idle":"2024-11-15T18:41:18.753678Z","shell.execute_reply.started":"2024-11-15T18:41:18.736386Z","shell.execute_reply":"2024-11-15T18:41:18.752873Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo mô hình và chuyển nó lên GPU\nmodel = CNNModel().to('cuda')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.754817Z","iopub.execute_input":"2024-11-15T18:41:18.75516Z","iopub.status.idle":"2024-11-15T18:41:18.994723Z","shell.execute_reply.started":"2024-11-15T18:41:18.755126Z","shell.execute_reply":"2024-11-15T18:41:18.993937Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH_SIZE = 16\nNUM_EPOCHS = 200\nLR = 1e-4","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:41:18.99715Z","iopub.execute_input":"2024-11-15T18:41:18.997444Z","iopub.status.idle":"2024-11-15T18:41:19.002948Z","shell.execute_reply.started":"2024-11-15T18:41:18.997418Z","shell.execute_reply":"2024-11-15T18:41:19.002259Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_data_loaders(fold: int, batch_size: int):\n    # Get the data splits for the given fold\n    train_data, val_data = AbdominalData('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv', current_fold=fold).get_splits()\n    \n    # Create and return train and validation data loaders\n    return (\n        DataLoader(train_data, batch_size=batch_size, shuffle=True),\n        DataLoader(val_data, batch_size=batch_size, shuffle=False)\n    )\n\n# Now, to get the data loaders for a specific fold, you can call:\ntrain_dataloader_0, val_dataloader_0 = get_data_loaders(0, BATCH_SIZE)\ntrain_dataloader_1, val_dataloader_1 = get_data_loaders(1, BATCH_SIZE)\ntrain_dataloader_2, val_dataloader_2 = get_data_loaders(2, BATCH_SIZE)\ntrain_dataloader_3, val_dataloader_3 = get_data_loaders(3, BATCH_SIZE)\ntrain_dataloader_4, val_dataloader_4 = get_data_loaders(4, BATCH_SIZE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T18:41:19.004193Z","iopub.execute_input":"2024-11-15T18:41:19.004467Z","iopub.status.idle":"2024-11-15T18:43:37.123708Z","shell.execute_reply.started":"2024-11-15T18:41:19.004443Z","shell.execute_reply":"2024-11-15T18:43:37.122666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_0, val_data_0 = AbdominalData('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv', current_fold=0).get_splits()\ntrain_data_1, val_data_1 = AbdominalData('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv', current_fold=1).get_splits()\ntrain_data_2, val_data_2 = AbdominalData('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv', current_fold=2).get_splits()\ntrain_data_3, val_data_3 = AbdominalData('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv', current_fold=3).get_splits()\ntrain_data_4, val_data_4 = AbdominalData('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv', current_fold=4).get_splits()\n\ntrain_dataloader_0 = DataLoader(train_data_0,batch_size = BATCH_SIZE, shuffle = True)\nval_dataloader_0 = DataLoader(val_data_0,batch_size = BATCH_SIZE, shuffle = False)\ntrain_dataloader_1 = DataLoader(train_data_1,batch_size = BATCH_SIZE, shuffle = True)\nval_dataloader_1 = DataLoader(val_data_1,batch_size = BATCH_SIZE, shuffle = False)\ntrain_dataloader_2 = DataLoader(train_data_2,batch_size = BATCH_SIZE, shuffle = True)\nval_dataloader_2 = DataLoader(val_data_2,batch_size = BATCH_SIZE, shuffle = False)\ntrain_dataloader_3 = DataLoader(train_data_3,batch_size = BATCH_SIZE, shuffle = True)\nval_dataloader_3 = DataLoader(val_data_3,batch_size = BATCH_SIZE, shuffle = False)\ntrain_dataloader_4 = DataLoader(train_data_4,batch_size = BATCH_SIZE, shuffle = True)\nval_dataloader_4 = DataLoader(val_data_4,batch_size = BATCH_SIZE, shuffle = False)","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:43:37.124938Z","iopub.execute_input":"2024-11-15T18:43:37.125237Z","iopub.status.idle":"2024-11-15T18:44:35.236552Z","shell.execute_reply.started":"2024-11-15T18:43:37.125208Z","shell.execute_reply":"2024-11-15T18:44:35.235559Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(), lr = LR)\nbce_b = nn.BCEWithLogitsLoss(pos_weight = torch.tensor([2.0]).to('cuda'))\nbce_e = nn.BCEWithLogitsLoss(pos_weight = torch.tensor([4.0]).to('cuda'))\ncce = nn.CrossEntropyLoss(label_smoothing = 0.05, weight = torch.tensor([1.0, 2.0, 4.0]).to('cuda'))\nscheduler = ReduceLROnPlateau(optimizer, mode='min', patience=5, factor=0.5, verbose=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:44:35.237734Z","iopub.execute_input":"2024-11-15T18:44:35.238053Z","iopub.status.idle":"2024-11-15T18:44:35.247333Z","shell.execute_reply.started":"2024-11-15T18:44:35.238025Z","shell.execute_reply":"2024-11-15T18:44:35.246304Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# initialize metrics objects\ntrain_acc_bowel = MetricsCalculator('binary')\ntrain_acc_extravasation = MetricsCalculator('binary')\ntrain_acc_liver = MetricsCalculator('multi')\ntrain_acc_kidney = MetricsCalculator('multi')\ntrain_acc_spleen = MetricsCalculator('multi')\n\nval_acc_bowel = MetricsCalculator('binary')\nval_acc_extravasation = MetricsCalculator('binary')\nval_acc_liver = MetricsCalculator('multi')\nval_acc_kidney = MetricsCalculator('multi')\nval_acc_spleen = MetricsCalculator('multi')","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:44:35.248643Z","iopub.execute_input":"2024-11-15T18:44:35.249085Z","iopub.status.idle":"2024-11-15T18:44:35.260295Z","shell.execute_reply.started":"2024-11-15T18:44:35.249051Z","shell.execute_reply":"2024-11-15T18:44:35.259449Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prev_val_best_loss = float('inf')\n\ndataloaders = [(train_dataloader_0, val_dataloader_0),\n               (train_dataloader_1, val_dataloader_1),\n               (train_dataloader_2, val_dataloader_2), \n               (train_dataloader_3, val_dataloader_3),\n               (train_dataloader_4, val_dataloader_4)]\n\nfor epoch in range(NUM_EPOCHS):\n    \n    # training\n    model.train()\n    \n    train_loss = 0.0\n    val_loss = 0.0\n    \n    print(f'Epoch: [{epoch+1}/{NUM_EPOCHS}]')\n    \n    train_dataloader, val_dataloader = dataloaders[epoch%5]\n    \n    print(f'Fold: {epoch%5}')\n    \n    for batch_idx, batch_data in enumerate(tqdm(train_dataloader)):\n        \n        inputs = batch_data['image'].to('cuda')\n        bowel = batch_data['bowel'].to('cuda')\n        extravasation = batch_data['extravasation'].to('cuda')\n        liver = batch_data['liver'].to('cuda')\n        kidney = batch_data['kidney'].to('cuda')\n        spleen = batch_data['spleen'].to('cuda')\n        \n        optimizer.zero_grad()\n        b, e, k, l, s = model(inputs)\n        b_loss = bce_b(b, bowel.float())\n        e_loss = bce_e(e, extravasation.float())\n        l_loss = cce(l, liver)\n        k_loss = cce(k, kidney)\n        s_loss = cce(s, spleen)\n        \n        total_loss = b_loss + e_loss + l_loss + k_loss + s_loss\n        total_loss.backward()\n        \n        optimizer.step()\n        \n        # calculate training metrics\n        train_loss += total_loss.item()\n        train_acc_bowel.update(b, bowel)\n        train_acc_extravasation.update(e, extravasation)\n        train_acc_liver.update(l, liver)\n        train_acc_kidney.update(k, kidney)\n        train_acc_spleen.update(s, spleen)\n    \n    train_loss = train_loss/(batch_idx+1)\n    \n    # validation\n    model.eval()\n    running_loss = 0.0\n    \n    for batch_idx, batch_data in enumerate(tqdm(val_dataloader)):\n                                                \n        inputs = batch_data['image'].to('cuda')\n        bowel = batch_data['bowel'].to('cuda')\n        extravasation = batch_data['extravasation'].to('cuda')\n        liver = batch_data['liver'].to('cuda')\n        kidney = batch_data['kidney'].to('cuda')\n        spleen = batch_data['spleen'].to('cuda')\n\n        \n        b, e, k, l, s = model(inputs)\n        b_loss = bce_b(b, bowel.float())\n        e_loss = bce_e(e, extravasation.float())\n        l_loss = cce(l, liver)\n        k_loss = cce(k, kidney)\n        s_loss = cce(s, spleen)\n        \n        total_loss = b_loss + e_loss + l_loss + k_loss + s_loss\n        \n        # calculate validation metrics\n        val_loss += total_loss.item()\n        val_acc_bowel.update(b, bowel)\n        val_acc_extravasation.update(e, extravasation)\n        val_acc_liver.update(l, liver)\n        val_acc_kidney.update(k, kidney)\n        val_acc_spleen.update(s, spleen)\n    \n    \n    val_loss = val_loss/(batch_idx+1)\n    scheduler.step(val_loss)\n    \n    if val_loss < prev_val_best_loss:\n        prev_val_best_loss = val_loss\n        print(\"Validation Loss improved, Saving Model...\")\n        torch.save(model, f'efficientnet_b0_{val_loss:.3f}.pth')\n    \n    \n    \n    # accuracy and auc data\n    metrics_data = [\n                    [\"Bowel\", \n                        train_acc_bowel.compute_accuracy(),\n                        val_acc_bowel.compute_accuracy(),\n                        train_acc_bowel.compute_auc(),\n                        val_acc_bowel.compute_auc()],\n                    [\"Extravasation\", \n                        train_acc_extravasation.compute_accuracy(),\n                        val_acc_extravasation.compute_accuracy(),\n                        train_acc_extravasation.compute_auc(),\n                        val_acc_extravasation.compute_auc()],\n                    [\"Liver\", \n                        train_acc_liver.compute_accuracy(),\n                        val_acc_liver.compute_accuracy(),\n                        train_acc_liver.compute_auc(),\n                        val_acc_liver.compute_auc()],\n                    [\"Kidney\", \n                        train_acc_kidney.compute_accuracy(),\n                        val_acc_kidney.compute_accuracy(),\n                        train_acc_kidney.compute_auc(),\n                        val_acc_kidney.compute_auc()],\n                    [\"Spleen\", \n                        train_acc_spleen.compute_accuracy(),\n                        val_acc_spleen.compute_accuracy(),\n                        train_acc_spleen.compute_auc(),\n                        val_acc_spleen.compute_auc()]\n                ]\n    \n    # verbose\n    print('')\n    print(tabulate(metrics_data, headers=[\"\", \"Train Acc\", \"Val Acc\", \"Train AUC\", \"Val AUC\"]))\n    \n    print(f'\\nMean Train Loss: {train_loss:.3f}')\n    print(f'Mean Val Loss: {val_loss:.3f}\\n')\n    \n    #reset metrics\n    train_acc_bowel.reset()\n    train_acc_extravasation.reset()\n    train_acc_liver.reset()\n    train_acc_kidney.reset()\n    train_acc_spleen.reset()\n    val_acc_bowel.reset()\n    val_acc_extravasation.reset()\n    val_acc_liver.reset()\n    val_acc_kidney.reset()\n    val_acc_spleen.reset()","metadata":{"execution":{"iopub.status.busy":"2024-11-15T18:44:35.262651Z","iopub.execute_input":"2024-11-15T18:44:35.263132Z","iopub.status.idle":"2024-11-15T18:48:57.153536Z","shell.execute_reply.started":"2024-11-15T18:44:35.263106Z","shell.execute_reply":"2024-11-15T18:48:57.152656Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom sklearn.metrics import make_scorer, precision_score, recall_score, f1_score\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\nimport numpy as np\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (ở đây sử dụng RandomForest)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Định nghĩa các hàm tính Precision, Recall, F1 và Accuracy\ndef calculate_precision(model, X, y):\n    return precision_score(y, model.predict(X), average='weighted')\n\ndef calculate_recall(model, X, y):\n    return recall_score(y, model.predict(X), average='weighted')\n\ndef calculate_f1(model, X, y):\n    return f1_score(y, model.predict(X), average='weighted')\n\n# Tính toán các chỉ số qua cross-validation\naccuracies = cross_val_score(model, X, y, cv=kf, scoring='accuracy')\nprecisions = cross_val_score(model, X, y, cv=kf, scoring=make_scorer(calculate_precision, greater_is_better=True))\nrecalls = cross_val_score(model, X, y, cv=kf, scoring=make_scorer(calculate_recall, greater_is_better=True))\nf1_scores = cross_val_score(model, X, y, cv=kf, scoring=make_scorer(calculate_f1, greater_is_better=True))\n\n# Tính toán trung bình của các chỉ số\navg_accuracy = np.mean(accuracies)\navg_precision = np.mean(precisions)\navg_recall = np.mean(recalls)\navg_f1 = np.mean(f1_scores)\n\n# In kết quả\nprint(f\"Accuracy trung bình: {avg_accuracy * 100:.2f}%\")\nprint(f\"Precision trung bình: {avg_precision * 100:.2f}%\")\nprint(f\"Recall trung bình: {avg_recall * 100:.2f}%\")\nprint(f\"F1 Score trung bình: {avg_f1 * 100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T18:52:29.166199Z","iopub.execute_input":"2024-11-15T18:52:29.166589Z","iopub.status.idle":"2024-11-15T18:52:33.902303Z","shell.execute_reply.started":"2024-11-15T18:52:29.166559Z","shell.execute_reply":"2024-11-15T18:52:33.901349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, StratifiedKFold, cross_val_predict\nfrom sklearn.metrics import make_scorer, precision_score, recall_score, f1_score, log_loss\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\nimport numpy as np\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (ở đây sử dụng RandomForest)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Tính toán các chỉ số qua cross-validation\naccuracies = cross_val_score(model, X, y, cv=kf, scoring='accuracy')\nprecisions = cross_val_score(model, X, y, cv=kf, scoring='precision_weighted')\nrecalls = cross_val_score(model, X, y, cv=kf, scoring='recall_weighted')\nf1_scores = cross_val_score(model, X, y, cv=kf, scoring='f1_weighted')\n\n# Tính toán trung bình của các chỉ số\navg_accuracy = np.mean(accuracies)\navg_precision = np.mean(precisions)\navg_recall = np.mean(recalls)\navg_f1 = np.mean(f1_scores)\n\n# Tính toán Validation Loss (log loss) và Validation Accuracy\nval_losses = []\nval_accuracies = []\n\n# Dùng cross_val_predict để dự đoán lớp cho mỗi fold\nfor train_index, val_index in kf.split(X, y):\n    X_train, X_val = X[train_index], X[val_index]\n    y_train, y_val = y[train_index], y[val_index]\n    \n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_val)\n    \n    # Tính Accuracy và Log Loss cho từng fold\n    val_accuracy = accuracy_score(y_val, y_pred)\n    val_loss = log_loss(y_val, model.predict_proba(X_val))\n    \n    val_accuracies.append(val_accuracy)\n    val_losses.append(val_loss)\n\n# Tính trung bình Validation Accuracy và Validation Loss\navg_val_accuracy = np.mean(val_accuracies)\navg_val_loss = np.mean(val_losses)\n\n# In kết quả\nprint(f\"Accuracy trung bình: {avg_accuracy * 100:.4f}%\")\nprint(f\"Precision trung bình: {avg_precision * 100:.4f}%\")\nprint(f\"Recall trung bình: {avg_recall * 100:.4f}%\")\nprint(f\"F1 Score trung bình: {avg_f1 * 100:.4f}%\")\nprint(f\"Validation Accuracy trung bình: {avg_val_accuracy * 100:.4f}%\")\nprint(f\"Validation Loss trung bình: {avg_val_loss:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:54:16.691943Z","iopub.execute_input":"2024-11-15T19:54:16.69261Z","iopub.status.idle":"2024-11-15T19:54:21.805024Z","shell.execute_reply.started":"2024-11-15T19:54:16.69258Z","shell.execute_reply":"2024-11-15T19:54:21.804033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install matplotlib pillow\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T18:48:57.191096Z","iopub.execute_input":"2024-11-15T18:48:57.191705Z","iopub.status.idle":"2024-11-15T18:49:11.021759Z","shell.execute_reply.started":"2024-11-15T18:48:57.191677Z","shell.execute_reply":"2024-11-15T18:49:11.020511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Hàm hiển thị hình ảnh với nhãn\ndef show_images(dataloader, num_images=5):\n    images_shown = 0\n    plt.figure(figsize=(15, 10))\n\n    for batch_data in dataloader:\n        images = batch_data['image']\n        labels = {\n            'bowel': batch_data['bowel'],\n            'extravasation': batch_data['extravasation'],\n            'liver': batch_data['liver'],\n            'kidney': batch_data['kidney'],\n            'spleen': batch_data['spleen']\n        }\n        \n        for i in range(images.size(0)):\n            if images_shown >= num_images:\n                plt.show()\n                return\n            \n            img = images[i].permute(1, 2, 0).cpu().numpy()  # Đổi thứ tự kênh ảnh từ (C, H, W) thành (H, W, C) để hiển thị\n            \n            # Hiển thị hình ảnh và các nhãn tương ứng\n            plt.subplot(1, num_images, images_shown + 1)\n            plt.imshow(img, cmap='gray')  # cmap='gray' nếu ảnh là grayscale\n            plt.axis('off')\n            \n            # Lấy các nhãn cho ảnh hiện tại\n            title = \"\\n\".join([f\"{key}: {label[i].item()}\" for key, label in labels.items()])\n            plt.title(title, fontsize=8)\n            \n            images_shown += 1\n\n    plt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:17:17.274176Z","iopub.execute_input":"2024-11-15T19:17:17.27458Z","iopub.status.idle":"2024-11-15T19:17:17.284923Z","shell.execute_reply.started":"2024-11-15T19:17:17.274546Z","shell.execute_reply":"2024-11-15T19:17:17.283778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Gọi hàm hiển thị một batch từ dataloader\nshow_images(train_dataloader, num_images=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T18:49:11.041065Z","iopub.execute_input":"2024-11-15T18:49:11.04147Z","iopub.status.idle":"2024-11-15T18:49:12.056822Z","shell.execute_reply.started":"2024-11-15T18:49:11.041438Z","shell.execute_reply":"2024-11-15T18:49:12.055871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install seaborn matplotlib\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:35:05.76781Z","iopub.execute_input":"2024-11-15T19:35:05.768687Z","iopub.status.idle":"2024-11-15T19:35:17.146514Z","shell.execute_reply.started":"2024-11-15T19:35:05.768656Z","shell.execute_reply":"2024-11-15T19:35:17.145276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (ví dụ RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Danh sách lưu kết quả accuracy của mỗi fold\nfold_results = []\n\n# Thực hiện cross-validation\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Huấn luyện mô hình\n    model.fit(X_train, y_train)\n    \n    # Dự đoán và tính accuracy cho validation set\n    y_pred = model.predict(X_val)\n    fold_accuracy = accuracy_score(y_val, y_pred)\n    \n    # Lưu kết quả độ chính xác của fold\n    fold_results.append(fold_accuracy)\n\n# Kiểm tra dữ liệu của fold_results\nprint(fold_results)\n\n# Vẽ Boxplot từ các kết quả fold_results\nplt.figure(figsize=(8, 6))\nsns.boxplot(data=fold_results)\n\n# Thêm tiêu đề và nhãn trục\nplt.title('Boxplot of Accuracy Across Different Folds')\nplt.xlabel('Folds')\nplt.ylabel('Accuracy')\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:38:54.935955Z","iopub.execute_input":"2024-11-15T19:38:54.936631Z","iopub.status.idle":"2024-11-15T19:38:56.176259Z","shell.execute_reply.started":"2024-11-15T19:38:54.936599Z","shell.execute_reply":"2024-11-15T19:38:56.17538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (ví dụ RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Danh sách lưu kết quả accuracy của mỗi fold\ntrain_accuracies = []  # Lưu độ chính xác huấn luyện\nval_accuracies = []    # Lưu độ chính xác validation\n\n# Số lượng epoch (folds)\nepochs = 5\n\n# Thực hiện cross-validation và huấn luyện mô hình qua các epoch\nfor epoch, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Huấn luyện mô hình\n    model.fit(X_train, y_train)\n    \n    # Dự đoán và tính accuracy cho tập huấn luyện và tập validation\n    y_train_pred = model.predict(X_train)\n    y_val_pred = model.predict(X_val)\n    \n    # Tính độ chính xác cho tập huấn luyện và validation\n    train_accuracy = accuracy_score(y_train, y_train_pred)\n    val_accuracy = accuracy_score(y_val, y_val_pred)\n    \n    # Lưu kết quả vào danh sách\n    train_accuracies.append(train_accuracy)\n    val_accuracies.append(val_accuracy)\n\n# Vẽ đồ thị accuracy qua các epoch\nepochs_range = np.arange(1, epochs + 1)\n\nplt.figure(figsize=(8, 6))\n\n# Vẽ độ chính xác huấn luyện\nplt.plot(epochs_range, train_accuracies, label='Train Accuracy', marker='o', linestyle='-', color='blue')\n\n# Vẽ độ chính xác validation\nplt.plot(epochs_range, val_accuracies, label='Validation Accuracy', marker='o', linestyle='--', color='green')\n\n# Thêm tiêu đề và nhãn\nplt.title('Accuracy vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.xticks(epochs_range)  # Đảm bảo trục x có đủ các nhãn cho mỗi epoch\nplt.legend()\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:57:23.064327Z","iopub.execute_input":"2024-11-15T19:57:23.065352Z","iopub.status.idle":"2024-11-15T19:57:24.410779Z","shell.execute_reply.started":"2024-11-15T19:57:23.065314Z","shell.execute_reply":"2024-11-15T19:57:24.409848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import log_loss\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (ví dụ RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Danh sách lưu kết quả accuracy và loss của mỗi fold\ntrain_losses = []  # Lưu giá trị loss huấn luyện\nval_losses = []    # Lưu giá trị loss validation\n\n# Số lượng epoch (folds)\nepochs = 5\n\n# Thực hiện cross-validation và huấn luyện mô hình qua các epoch\nfor epoch, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Huấn luyện mô hình\n    model.fit(X_train, y_train)\n    \n    # Dự đoán và tính accuracy cho tập huấn luyện và tập validation\n    y_train_pred = model.predict(X_train)\n    y_val_pred = model.predict(X_val)\n    \n    # Tính loss cho tập huấn luyện và validation\n    train_loss = log_loss(y_train, model.predict_proba(X_train))  # Sử dụng log_loss cho huấn luyện\n    val_loss = log_loss(y_val, model.predict_proba(X_val))  # Sử dụng log_loss cho validation\n    \n    # Lưu kết quả vào danh sách\n    train_losses.append(train_loss)\n    val_losses.append(val_loss)\n\n# Vẽ đồ thị loss qua các epoch\nepochs_range = np.arange(1, epochs + 1)\n\nplt.figure(figsize=(8, 6))\n\n# Vẽ train loss\nplt.plot(epochs_range, train_losses, label='Train Loss', marker='o', linestyle='-', color='blue')\n\n# Vẽ validation loss\nplt.plot(epochs_range, val_losses, label='Validation Loss', marker='o', linestyle='--', color='green')\n\n# Thêm tiêu đề và nhãn\nplt.title('Loss vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\n\n# Đảm bảo trục x có đủ các nhãn cho mỗi epoch (1, 2, 3, ..., 5)\nplt.xticks(epochs_range)\n\n# Thêm chú thích cho đồ thị\nplt.legend()\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:42:29.291332Z","iopub.execute_input":"2024-11-15T19:42:29.291745Z","iopub.status.idle":"2024-11-15T19:42:30.718577Z","shell.execute_reply.started":"2024-11-15T19:42:29.291712Z","shell.execute_reply":"2024-11-15T19:42:30.717674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Lưu kết quả confusion matrix cho mỗi fold\nconfusion_matrices = []\n\n# Thực hiện cross-validation và huấn luyện mô hình qua các epoch\nfor epoch, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Huấn luyện mô hình\n    model.fit(X_train, y_train)\n    \n    # Dự đoán cho tập validation\n    y_val_pred = model.predict(X_val)\n    \n    # Tính ma trận nhầm lẫn cho tập validation\n    cm = confusion_matrix(y_val, y_val_pred)\n    confusion_matrices.append(cm)\n\n# Tính trung bình ma trận nhầm lẫn trên tất cả các fold\navg_confusion_matrix = np.mean(confusion_matrices, axis=0)\n\n# Vẽ ma trận nhầm lẫn\nplt.figure(figsize=(8, 6))\nsns.heatmap(avg_confusion_matrix, annot=True, fmt='g', cmap='Blues', xticklabels=data.target_names, yticklabels=data.target_names)\n\n# Thêm tiêu đề và nhãn trục\nplt.title('Average Confusion Matrix (Over All Folds)')\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:44:07.581971Z","iopub.execute_input":"2024-11-15T19:44:07.582366Z","iopub.status.idle":"2024-11-15T19:44:08.850158Z","shell.execute_reply.started":"2024-11-15T19:44:07.582335Z","shell.execute_reply":"2024-11-15T19:44:08.849207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.datasets import load_digits  # Dữ liệu MNIST hoặc dataset tương tự\nfrom sklearn.ensemble import RandomForestClassifier\n\n# Tải dữ liệu ví dụ (Digits dataset, có 10 lớp từ 0 đến 9)\nfrom sklearn.datasets import load_digits\ndata = load_digits()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Lưu kết quả confusion matrix cho mỗi fold\nconfusion_matrices = []\n\n# Thực hiện cross-validation và huấn luyện mô hình qua các epoch\nfor epoch, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Huấn luyện mô hình\n    model.fit(X_train, y_train)\n    \n    # Dự đoán cho tập validation\n    y_val_pred = model.predict(X_val)\n    \n    # Tính ma trận nhầm lẫn cho tập validation\n    cm = confusion_matrix(y_val, y_val_pred)\n    confusion_matrices.append(cm)\n\n# Tính trung bình ma trận nhầm lẫn trên tất cả các fold\navg_confusion_matrix = np.mean(confusion_matrices, axis=0)\n\n# Vẽ ma trận nhầm lẫn\nplt.figure(figsize=(8, 6))\nsns.heatmap(avg_confusion_matrix, annot=True, fmt='g', cmap='Blues', xticklabels=data.target_names, yticklabels=data.target_names)\n\n# Thêm tiêu đề và nhãn trục\nplt.title('Average Confusion Matrix (Over All Folds)')\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:44:55.977914Z","iopub.execute_input":"2024-11-15T19:44:55.978276Z","iopub.status.idle":"2024-11-15T19:44:58.822485Z","shell.execute_reply.started":"2024-11-15T19:44:55.978247Z","shell.execute_reply":"2024-11-15T19:44:58.821557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.datasets import load_digits\nfrom sklearn.ensemble import RandomForestClassifier\n\n# Tải dữ liệu ví dụ (Digits dataset, có 10 lớp từ 0 đến 9)\ndata = load_digits()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Danh sách lưu kết quả của từng metrics\ntrain_accuracies = []\nval_accuracies = []\ntrain_precisions = []\nval_precisions = []\ntrain_recalls = []\nval_recalls = []\ntrain_f1_scores = []\nval_f1_scores = []\n\n# Số lượng epoch (folds)\nepochs = 5\n\n# Thực hiện cross-validation và huấn luyện mô hình qua các epoch\nfor epoch, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # Huấn luyện mô hình\n    model.fit(X_train, y_train)\n    \n    # Dự đoán cho tập huấn luyện và tập validation\n    y_train_pred = model.predict(X_train)\n    y_val_pred = model.predict(X_val)\n    \n    # Tính các metrics cho tập huấn luyện và tập validation\n    train_accuracy = accuracy_score(y_train, y_train_pred)\n    val_accuracy = accuracy_score(y_val, y_val_pred)\n    \n    train_precision = precision_score(y_train, y_train_pred, average='weighted')\n    val_precision = precision_score(y_val, y_val_pred, average='weighted')\n    \n    train_recall = recall_score(y_train, y_train_pred, average='weighted')\n    val_recall = recall_score(y_val, y_val_pred, average='weighted')\n    \n    train_f1 = f1_score(y_train, y_train_pred, average='weighted')\n    val_f1 = f1_score(y_val, y_val_pred, average='weighted')\n    \n    # Lưu kết quả vào các danh sách\n    train_accuracies.append(train_accuracy)\n    val_accuracies.append(val_accuracy)\n    \n    train_precisions.append(train_precision)\n    val_precisions.append(val_precision)\n    \n    train_recalls.append(train_recall)\n    val_recalls.append(val_recall)\n    \n    train_f1_scores.append(train_f1)\n    val_f1_scores.append(val_f1)\n\n# Tạo các biểu đồ cho từng metrics\nepochs_range = np.arange(1, epochs + 1)\n\n# Vẽ biểu đồ cho accuracy\nplt.figure(figsize=(10, 6))\nplt.plot(epochs_range, train_accuracies, label='Train Accuracy', marker='o', color='blue')\nplt.plot(epochs_range, val_accuracies, label='Validation Accuracy', marker='o', color='green')\nplt.title('Accuracy vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.xticks(epochs_range)\nplt.legend()\nplt.show()\n\n# Vẽ biểu đồ cho precision\nplt.figure(figsize=(10, 6))\nplt.plot(epochs_range, train_precisions, label='Train Precision', marker='o', color='blue')\nplt.plot(epochs_range, val_precisions, label='Validation Precision', marker='o', color='green')\nplt.title('Precision vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Precision')\nplt.xticks(epochs_range)\nplt.legend()\nplt.show()\n\n# Vẽ biểu đồ cho recall\nplt.figure(figsize=(10, 6))\nplt.plot(epochs_range, train_recalls, label='Train Recall', marker='o', color='blue')\nplt.plot(epochs_range, val_recalls, label='Validation Recall', marker='o', color='green')\nplt.title('Recall vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Recall')\nplt.xticks(epochs_range)\nplt.legend()\nplt.show()\n\n# Vẽ biểu đồ cho F1 Score\nplt.figure(figsize=(10, 6))\nplt.plot(epochs_range, train_f1_scores, label='Train F1 Score', marker='o', color='blue')\nplt.plot(epochs_range, val_f1_scores, label='Validation F1 Score', marker='o', color='green')\nplt.title('F1 Score vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('F1 Score')\nplt.xticks(epochs_range)\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:46:59.558492Z","iopub.execute_input":"2024-11-15T19:46:59.559255Z","iopub.status.idle":"2024-11-15T19:47:03.037413Z","shell.execute_reply.started":"2024-11-15T19:46:59.559218Z","shell.execute_reply":"2024-11-15T19:47:03.036376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nimport numpy as np\n\n# Giả sử y_true và y_pred là danh sách nhãn thật và nhãn dự đoán\n# Ví dụ dữ liệu giả cho nhãn thật và nhãn dự đoán\ny_true = [\n    'bowel_healthy', 'bowel_injury', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n    'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy',\n    'liver_low', 'kidney_low', 'liver_healthy', 'bowel_injury', 'liver_low'\n]  # Các nhãn thật\n\ny_pred = [\n    'bowel_healthy', 'bowel_injury', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n    'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy',\n    'liver_low', 'kidney_low', 'liver_healthy', 'bowel_injury', 'liver_healthy'\n]  # Các nhãn dự đoán (giả sử giống với y_true)\n\n# Danh sách các lớp (labels)\nlabels = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low'\n]\n\n# Tính toán ma trận nhầm lẫn\ncm = confusion_matrix(y_true, y_pred, labels=labels)\n\n# Vẽ ma trận nhầm lẫn\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=labels, yticklabels=labels, cbar=False)\n\n# Thêm tiêu đề và nhãn trục\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:49:15.179812Z","iopub.execute_input":"2024-11-15T19:49:15.180703Z","iopub.status.idle":"2024-11-15T19:49:15.617386Z","shell.execute_reply.started":"2024-11-15T19:49:15.180669Z","shell.execute_reply":"2024-11-15T19:49:15.616378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.datasets import load_iris\nfrom sklearn.metrics import accuracy_score\n\n# Giả sử bạn đã huấn luyện một mô hình và có các giá trị loss (ví dụ, train_loss, val_loss)\ntrain_losses = []\nval_losses = []\n\n# Dữ liệu giả lập (ví dụ: Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Đổi dữ liệu thành torch tensors\nX = torch.tensor(X, dtype=torch.float32)\ny = torch.tensor(y, dtype=torch.long)\n\n# Khởi tạo mô hình, loss function và optimizer (ví dụ sử dụng một mô hình đơn giản)\nmodel = nn.Sequential(\n    nn.Linear(X.shape[1], 50),\n    nn.ReLU(),\n    nn.Linear(50, len(set(y.numpy())))\n)\n\nloss_fn = nn.CrossEntropyLoss()\noptimizer = optim.SGD(model.parameters(), lr=0.01)\n\n# Khởi tạo StratifiedKFold\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Thực hiện cross-validation\nfor epoch in range(5):  # Giả sử 5 epoch\n    for train_idx, val_idx in kf.split(X.numpy(), y.numpy()):\n        # Chia dữ liệu thành tập huấn luyện và tập validation\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n        \n        # Huấn luyện mô hình\n        model.train()\n        optimizer.zero_grad()\n        output = model(X_train)\n        loss = loss_fn(output, y_train)\n        \n        # Cập nhật trọng số\n        loss.backward()\n        optimizer.step()\n        \n        # Lưu train loss\n        train_losses.append(loss.item())\n        \n        # Validation loss\n        model.eval()\n        with torch.no_grad():\n            val_output = model(X_val)\n            val_loss = loss_fn(val_output, y_val)\n            val_losses.append(val_loss.item())\n\n# Vẽ đồ thị\nepochs_range = np.arange(1, len(train_losses) + 1)\n\nplt.figure(figsize=(10, 6))\n\n# Vẽ train loss\nplt.plot(epochs_range, train_losses, label='Train Loss', marker='o', linestyle='-', color='blue')\n\n# Vẽ validation loss\nplt.plot(epochs_range, val_losses, label='Validation Loss', marker='o', linestyle='--', color='green')\n\n# Thêm tiêu đề và nhãn\nplt.title('Train Loss vs Validation Loss Across Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.xticks(epochs_range)  # Đảm bảo trục x có đủ các nhãn cho mỗi epoch\nplt.legend()\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T19:53:28.651714Z","iopub.execute_input":"2024-11-15T19:53:28.652588Z","iopub.status.idle":"2024-11-15T19:53:29.112012Z","shell.execute_reply.started":"2024-11-15T19:53:28.652552Z","shell.execute_reply":"2024-11-15T19:53:29.111047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.datasets import load_iris\nfrom sklearn.ensemble import RandomForestClassifier\n\n# Tải dữ liệu ví dụ (Iris dataset)\ndata = load_iris()\nX = data.data\ny = data.target\n\n# Khởi tạo mô hình (RandomForestClassifier)\nmodel = RandomForestClassifier()\n\n# Khởi tạo StratifiedKFold (chọn k=5 để phân chia dữ liệu)\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Danh sách lưu kết quả accuracy của mỗi fold\ntrain_accuracies = []  # Lưu độ chính xác huấn luyện\nval_accuracies = []    # Lưu độ chính xác validation\n\n# Số lượng epoch (200 epoch cho cross-validation)\nepochs = 200\n\n# Thực hiện cross-validation và huấn luyện mô hình qua các epoch\nfor epoch in range(epochs):\n    for train_idx, val_idx in kf.split(X, y):\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n        \n        # Huấn luyện mô hình\n        model.fit(X_train, y_train)\n        \n        # Dự đoán và tính accuracy cho tập huấn luyện và tập validation\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n        \n        # Tính độ chính xác cho tập huấn luyện và validation\n        train_accuracy = accuracy_score(y_train, y_train_pred)\n        val_accuracy = accuracy_score(y_val, y_val_pred)\n        \n        # Lưu kết quả vào danh sách\n        train_accuracies.append(train_accuracy)\n        val_accuracies.append(val_accuracy)\n\n# Vẽ đồ thị accuracy qua các epoch\nepochs_range = np.arange(1, epochs + 1)\n\nplt.figure(figsize=(10, 6))\n\n# Vẽ train accuracy\nplt.plot(epochs_range, train_accuracies, label='Train Accuracy', marker='o', linestyle='-', color='blue')\n\n# Vẽ validation accuracy\nplt.plot(epochs_range, val_accuracies, label='Validation Accuracy', marker='o', linestyle='--', color='green')\n\n# Thêm tiêu đề và nhãn\nplt.title('Accuracy vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\n\n# Thiết lập các mốc trên trục X cách nhau 25 đơn vị\nplt.xticks(np.arange(1, epochs + 1, step=25))  # Các mốc cách nhau 25 đơn vị\n\n# Thêm chú thích cho đồ thị\nplt.legend()\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}