{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":5838650,"sourceType":"datasetVersion","datasetId":3356068},{"sourceId":6267500,"sourceType":"datasetVersion","datasetId":3581068},{"sourceId":6367091,"sourceType":"datasetVersion","datasetId":3668143},{"sourceId":6437750,"sourceType":"datasetVersion","datasetId":3715314},{"sourceId":6440340,"sourceType":"datasetVersion","datasetId":3716919},{"sourceId":6501155,"sourceType":"datasetVersion","datasetId":3758078},{"sourceId":6504435,"sourceType":"datasetVersion","datasetId":3760213},{"sourceId":6524344,"sourceType":"datasetVersion","datasetId":3771912},{"sourceId":6533062,"sourceType":"datasetVersion","datasetId":3776927},{"sourceId":6605331,"sourceType":"datasetVersion","datasetId":3695414},{"sourceId":6642094,"sourceType":"datasetVersion","datasetId":3834301},{"sourceId":6666397,"sourceType":"datasetVersion","datasetId":3846730},{"sourceId":6670204,"sourceType":"datasetVersion","datasetId":3848762},{"sourceId":6673109,"sourceType":"datasetVersion","datasetId":3850050},{"sourceId":6673115,"sourceType":"datasetVersion","datasetId":3850054},{"sourceId":6675114,"sourceType":"datasetVersion","datasetId":3851220},{"sourceId":6678504,"sourceType":"datasetVersion","datasetId":3852993},{"sourceId":6678520,"sourceType":"datasetVersion","datasetId":3853000},{"sourceId":6679801,"sourceType":"datasetVersion","datasetId":3853474},{"sourceId":6680756,"sourceType":"datasetVersion","datasetId":3853894},{"sourceId":6681262,"sourceType":"datasetVersion","datasetId":3854158},{"sourceId":6681310,"sourceType":"datasetVersion","datasetId":3854197},{"sourceId":6681564,"sourceType":"datasetVersion","datasetId":3854378},{"sourceId":6694288,"sourceType":"datasetVersion","datasetId":3859591},{"sourceId":6694678,"sourceType":"datasetVersion","datasetId":3859765},{"sourceId":6695864,"sourceType":"datasetVersion","datasetId":3860267},{"sourceId":7432254,"sourceType":"datasetVersion","datasetId":4325089},{"sourceId":7665407,"sourceType":"datasetVersion","datasetId":4470305},{"sourceId":6523471,"sourceType":"datasetVersion","datasetId":3771357},{"sourceId":146649063,"sourceType":"kernelVersion"},{"sourceId":146650289,"sourceType":"kernelVersion"}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install nibabel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:56:25.683794Z","iopub.execute_input":"2024-12-01T17:56:25.684071Z","iopub.status.idle":"2024-12-01T17:56:34.92665Z","shell.execute_reply.started":"2024-12-01T17:56:25.684046Z","shell.execute_reply":"2024-12-01T17:56:34.925406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport re\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom as dicom\nimport nibabel as nib\nfrom tqdm import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:56:34.928668Z","iopub.execute_input":"2024-12-01T17:56:34.92895Z","iopub.status.idle":"2024-12-01T17:56:36.018961Z","shell.execute_reply.started":"2024-12-01T17:56:34.928926Z","shell.execute_reply":"2024-12-01T17:56:36.018289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nimport random\nimport re\n\nfrom tqdm import tqdm\n\nimport pydicom as dicom\nimport nibabel as nib","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DS_RATE = 2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:56:36.019985Z","iopub.execute_input":"2024-12-01T17:56:36.020685Z","iopub.status.idle":"2024-12-01T17:56:36.024875Z","shell.execute_reply.started":"2024-12-01T17:56:36.020651Z","shell.execute_reply":"2024-12-01T17:56:36.024108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DS_RATE = 2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn import Transformer\n\ninput_shape = (128, 128, 128)  # Depth x Height x Width\nnum_classes = 14  # Number of classes for classification\n\nclass Transformer3DClassifier(nn.Module):\n    def __init__(self, input_shape, num_classes, num_layers=12, d_model=64, nhead=8, dim_feedforward=2048, dropout=0.1):\n        super(Transformer3DClassifier, self).__init__()\n\n        # Initialize d_model\n        self.d_model = d_model\n        # Calculate the input size for the transformer\n        d_in = input_shape[0] * input_shape[1] * input_shape[2]  # Depth x Height x Width\n        self.embedding = nn.Linear(d_in, d_model)\n        self.transformer = Transformer(\n            d_model=d_model,\n            nhead=nhead,\n            num_encoder_layers=num_layers,\n            dim_feedforward=dim_feedforward,\n            dropout=dropout\n        )\n        self.fc = nn.Linear(d_model, num_classes)\n\n    def forward(self, x):\n        # Flatten the input and apply linear embedding\n        x = x.view(x.size(0), -1)\n        print(\"Before embedding x.shape is \", x.shape)\n        x = self.embedding(x)\n        print(\"After embedding x.shape is \", x.shape)\n        # Reshape to add a third dimension (seq_len)\n        x = x.unsqueeze(0)\n        print(\"x shape after unsqueeze\", x.shape)\n        # Create a dummy target tensor (you can adjust its size if needed)\n        tgt = torch.zeros(1, x.size(1), self.d_model).to(x.device)\n        print(\"tgt shape\", tgt.shape)\n        # Transformer encoder\n        output = self.transformer(x, tgt)\n        print(\"Output shape after transformer\", output.shape)\n        logits = self.fc(output)\n        logits = logits.unsqueeze(0)\n        return logits\n# Define tensors for input and hidden layer sizes\ninput_size = 32  # Adjust the dimensions as needed\nhidden_size = 16  # Adjust the dimensions as needed\n\nclass Custom3DViTModel(nn.Module):\n    def __init__(self, in_channels, num_classes, batch_size):\n        super(Custom3DViTModel, self).__init__()\n        self.batch_size = batch_size\n        self.num_classes = num_classes\n        self.vit_backbone = Transformer3DClassifier(\n            input_shape,\n            num_classes\n        )\n        self.classification_head = nn.Sequential(\n            nn.Linear(self.vit_backbone.d_model, num_classes)\n        )\n\n    def forward(self, x):\n        print(\"x shape and segmentation_mask shape\", x.shape)\n        features = self.vit_backbone(x)\n        print(\"features shape\", features.shape)\n        classification_output = features.view(self.batch_size, self.num_classes)\n        print(\"classification_output\", classification_output.shape)\n        return classification_output\n\n\n# Define the model\nmodel_class = Custom3DViTModel(1, 14, 32)\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")  # Choose the appropriate device\nmodel_class=model_class.to(device)\n# Create sample input tensors (modify this according to your data)\nbatch_images = torch.randn(32, 1, 128, 128, 128)  # Example input shape\nbatch_images = batch_images.to(device)  # Move input tensors to device\nprint(model_class)\n# Forward pass\n# classification_outputs = model(batch_images)\n# print(classification_outputs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:56:36.027716Z","iopub.execute_input":"2024-12-01T17:56:36.02805Z","iopub.status.idle":"2024-12-01T17:56:41.194306Z","shell.execute_reply.started":"2024-12-01T17:56:36.02802Z","shell.execute_reply":"2024-12-01T17:56:41.193304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install einops\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:56:41.195376Z","iopub.execute_input":"2024-12-01T17:56:41.195799Z","iopub.status.idle":"2024-12-01T17:56:49.591429Z","shell.execute_reply.started":"2024-12-01T17:56:41.195776Z","shell.execute_reply":"2024-12-01T17:56:49.590335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_3D_scans(folder, downsample_rate=1): \n    filenames = os.listdir(folder)\n    filenames = [int(filename.split('.')[0]) for filename in filenames]\n    filenames = sorted(filenames)\n    filenames = [str(filename) + '.dcm' for filename in filenames]\n        \n    volume = []\n    for filename in tqdm(filenames[::downsample_rate]):\n        filepath = os.path.join(folder, filename)\n        ds = dicom.dcmread(filepath)\n        image = ds.pixel_array\n        \n        # find rescale params\n        if (\"RescaleIntercept\" in ds) and (\"RescaleSlope\" in ds):\n            intercept = float(ds.RescaleIntercept)\n            slope = float(ds.RescaleSlope)\n    \n        # find clipping params\n        center = int(ds.WindowCenter)\n        width = int(ds.WindowWidth)\n        low = center - width / 2\n        high = center + width / 2    \n        \n        \n        image = (image * slope) + intercept\n        image = np.clip(image, low, high)\n\n        image = (image / np.max(image) * 255).astype(np.int16)\n        image = image[::downsample_rate, ::downsample_rate]\n        volume.append( image )\n    \n    volume = np.stack(volume, axis=0)\n    return volume\n\n\ndef create_3D_segmentations(filepath, downsample_rate=1):\n    img = nib.load(filepath).get_fdata()\n    img = np.transpose(img, [1, 0, 2])\n    img = np.rot90(img, 1, (1,2))\n    img = img[::-1,:,:]\n    img = np.transpose(img, [1, 0, 2])\n    img = img[::downsample_rate, ::downsample_rate, ::downsample_rate]\n    return img\n\n\n\nfilepath = '/kaggle/input/rsna-2023-abdominal-trauma-detection/segmentations/21057.nii'\nvolume_seg = create_3D_segmentations(filepath, downsample_rate=DS_RATE)\nprint(f'3D segmentation file shape: {volume_seg.shape}')\n\nfilepath = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057'\nvolume = create_3D_scans(filepath, downsample_rate=DS_RATE)\nprint(f'3D Image file shape: {volume.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:31:25.953299Z","iopub.execute_input":"2024-12-01T18:31:25.953671Z","iopub.status.idle":"2024-12-01T18:31:42.477967Z","shell.execute_reply.started":"2024-12-01T18:31:25.953638Z","shell.execute_reply":"2024-12-01T18:31:42.477148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import các thư viện cần thiết\nimport os\nimport numpy as np\nimport pandas as pd\nimport nibabel as nib\nfrom scipy.ndimage import zoom\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import KFold\nfrom torchvision import transforms\nfrom tqdm import tqdm\n\n# 1. Định Nghĩa CustomDataset\nclass CustomDataset(Dataset):\n    def __init__(self, file_paths, labels, transform=None, desired_shape=(128, 128, 128)):\n        \"\"\"\n        Args:\n            file_paths (list): Danh sách đường dẫn đến các file .npy đã được tiền xử lý.\n            labels (numpy.ndarray): Mảng nhãn tương ứng với các file.\n            transform (callable, optional): Biến đổi sẽ được áp dụng lên ảnh.\n            desired_shape (tuple, optional): Kích thước mong muốn của ảnh sau khi thay đổi kích thước.\n        \"\"\"\n        self.file_paths = file_paths\n        self.labels = labels\n        self.transform = transform\n        self.desired_shape = desired_shape\n\n    def __len__(self):\n        return len(self.file_paths)\n\n    def __getitem__(self, idx):\n        file_path = self.file_paths[idx]\n        label = self.labels[idx]\n\n        # Tải ảnh đã được tiền xử lý từ file .npy\n        image = np.load(file_path)\n        \n        # Nếu cần thay đổi kích thước (mặc định đã được thay đổi trong tiền xử lý)\n        if self.desired_shape and image.shape != self.desired_shape:\n            factors = (\n                self.desired_shape[0] / image.shape[0],\n                self.desired_shape[1] / image.shape[1],\n                self.desired_shape[2] / image.shape[2]\n            )\n            image = zoom(image, factors, order=3)\n            image = image.astype(np.float32)\n\n        # Chuyển đổi numpy array thành tensor và thêm chiều kênh\n        image = torch.from_numpy(image).float().unsqueeze(0)  # [1, D, H, W]\n\n        # Áp dụng biến đổi nếu có\n        if self.transform:\n            image = self.transform(image)\n\n        # Chuyển đổi nhãn thành tensor\n        label = torch.from_numpy(label).float()\n\n        return image, label\n\nclass Simple3DModel(nn.Module):\n    def __init__(self, num_classes=14, input_shape=(128, 128, 128)):\n        super(Simple3DModel, self).__init__()\n        self.conv1 = nn.Conv3d(in_channels=1, out_channels=16, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool3d(kernel_size=2)\n        self.conv2 = nn.Conv3d(in_channels=16, out_channels=32, kernel_size=3, padding=1)\n        self.conv3 = nn.Conv3d(in_channels=32, out_channels=64, kernel_size=3, padding=1)\n        self.conv4 = nn.Conv3d(in_channels=64, out_channels=128, kernel_size=3, padding=1)\n        self.dropout = nn.Dropout(p=0.5)\n        \n        \n        # Tính kích thước sau pooling\n        reduced_shape = (input_shape[0] // 16, input_shape[1] // 16, input_shape[2] // 16)\n        self.flatten_size = 128 * reduced_shape[0] * reduced_shape[1] * reduced_shape[2]\n        \n        # Fully Connected Layers\n        self.fc1 = nn.Linear(self.flatten_size, 256)\n        self.bn1 = nn.BatchNorm1d(256)\n        self.fc1_extra = nn.Linear(256, 128)\n        self.bn2 = nn.BatchNorm1d(128)\n        self.fc2 = nn.Linear(128, num_classes)\n        self.dropout_fc = nn.Dropout(p=0.5)  # Dropout cho tầng fully-connected\n        \n    def forward(self, x):\n        x = torch.relu(self.conv1(x))  # [batch, 16, 128, 128, 128]\n        x = self.pool(x)               # [batch, 16, 64, 64, 64]\n        x = torch.relu(self.conv2(x))  # [batch, 32, 64, 64, 64]\n        x = self.pool(x)               # [batch, 32, 32, 32, 32]\n        x = torch.relu(self.conv3(x))  # [batch, 64, 32, 32, 32]\n        x = self.pool(x)               # [batch, 64, 16, 16, 16]\n        x = torch.relu(self.conv4(x))  # [batch, 128, 16, 16, 16]\n        x = self.pool(x)               # [batch, 128, 8, 8, 8]\n        x = self.dropout(x)            # [batch, 128, 8, 8, 8]\n        \n        x = x.view(x.size(0), -1)      # Flatten: [batch, 128*8*8*8]\n        x = torch.relu(self.bn1(self.fc1(x)))  # [batch, 256]\n        x = torch.relu(self.bn2(self.fc1_extra(x)))  # [batch, 128]\n        x = self.fc2(x)                # [batch, num_classes]\n        return x\n\n\n\n# 3. Định Nghĩa train_validate_fold với tính toán các chỉ số bổ sung\ndef train_validate_fold(fold, train_loader, val_loader, model, criterion, optimizer, scheduler, device, num_epochs, patience=5):\n    \"\"\"\n    Huấn luyện và đánh giá mô hình cho một fold cụ thể, bao gồm các chỉ số đánh giá.\n    \"\"\"\n    metrics = {\n        'train_loss': [], 'val_loss': [],\n        'train_accuracy': [], 'val_accuracy': [],\n        'sensitivity': [], 'specificity': [], 'f1_score': []\n    }\n\n    best_val_loss = float('inf')\n    epochs_no_improve = 0  # Đếm số epoch không cải thiện\n    best_model_path = f\"best_model_fold_{fold + 1}.pth\"  # Tên file lưu mô hình tốt nhất\n\n    for epoch in tqdm(range(num_epochs), desc=f\"Fold {fold + 1} Training\"):\n        # Giai đoạn Huấn luyện\n        model.train()\n        running_loss = 0.0\n        correct = 0\n        total = 0\n\n        TP, FP, TN, FN = 0, 0, 0, 0  # True Positives, False Positives, True Negatives, False Negatives\n\n        for images, labels in train_loader:\n            images, labels = images.to(device), labels.to(device)\n\n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n\n            running_loss += loss.item()\n\n            preds = (torch.sigmoid(outputs) > 0.5).float()\n            correct += (preds == labels).sum().item()\n            total += labels.numel()\n\n            TP += ((preds == 1) & (labels == 1)).sum().item()\n            FP += ((preds == 1) & (labels == 0)).sum().item()\n            TN += ((preds == 0) & (labels == 0)).sum().item()\n            FN += ((preds == 0) & (labels == 1)).sum().item()\n\n        train_loss = running_loss / len(train_loader)\n        train_accuracy = correct / total\n        sensitivity = TP / (TP + FN) if (TP + FN) > 0 else 0\n        specificity = TN / (TN + FP) if (TN + FP) > 0 else 0\n        f1_score = 2 * TP / (2 * TP + FP + FN) if (2 * TP + FP + FN) > 0 else 0\n\n        metrics['train_loss'].append(train_loss)\n        metrics['train_accuracy'].append(train_accuracy)\n        metrics['sensitivity'].append(sensitivity)\n        metrics['specificity'].append(specificity)\n        metrics['f1_score'].append(f1_score)\n\n        # Giai đoạn Đánh giá\n        model.eval()\n        val_loss = 0.0\n        val_correct = 0\n        val_total = 0\n\n        val_TP, val_FP, val_TN, val_FN = 0, 0, 0, 0\n\n        with torch.no_grad():\n            for val_images, val_labels in val_loader:\n                val_images, val_labels = val_images.to(device), val_labels.to(device)\n                val_outputs = model(val_images)\n                loss = criterion(val_outputs, val_labels)\n                val_loss += loss.item()\n\n                val_preds = (torch.sigmoid(val_outputs) > 0.5).float()\n                val_correct += (val_preds == val_labels).sum().item()\n                val_total += val_labels.numel()\n\n                val_TP += ((val_preds == 1) & (val_labels == 1)).sum().item()\n                val_FP += ((val_preds == 1) & (val_labels == 0)).sum().item()\n                val_TN += ((val_preds == 0) & (val_labels == 0)).sum().item()\n                val_FN += ((val_preds == 0) & (val_labels == 1)).sum().item()\n\n        val_loss /= len(val_loader)\n        val_accuracy = val_correct / val_total\n        val_sensitivity = val_TP / (val_TP + val_FN) if (val_TP + val_FN) > 0 else 0\n        val_specificity = val_TN / (val_TN + val_FP) if (val_TN + val_FP) > 0 else 0\n        val_f1_score = 2 * val_TP / (2 * val_TP + val_FP + val_FN) if (2 * val_TP + val_FP + val_FN) > 0 else 0\n\n        metrics['val_loss'].append(val_loss)\n        metrics['val_accuracy'].append(val_accuracy)\n        metrics['sensitivity'].append(val_sensitivity)\n        metrics['specificity'].append(val_specificity)\n        metrics['f1_score'].append(val_f1_score)\n\n        print(f\"Fold {fold + 1}, Epoch [{epoch + 1}/{num_epochs}] | \"\n              f\"Train Loss: {train_loss:.4f}, Train Acc: {train_accuracy:.4f}, \"\n              f\"Sens: {sensitivity:.4f}, Spec: {specificity:.4f}, F1: {f1_score:.4f} | \"\n              f\"Val Loss: {val_loss:.4f}, Val Acc: {val_accuracy:.4f}, \"\n              f\"Sens: {val_sensitivity:.4f}, Spec: {val_specificity:.4f}, F1: {val_f1_score:.4f}\")\n\n        # Kiểm tra cải thiện val_loss\n        if val_loss < best_val_loss:\n            best_val_loss = val_loss\n            epochs_no_improve = 0\n            torch.save(model.state_dict(), best_model_path)  # Lưu mô hình tốt nhất\n        else:\n            epochs_no_improve += 1\n\n        # Scheduler bước dựa trên val_loss\n        scheduler.step(val_loss)\n\n        # Dừng sớm nếu không cải thiện sau 'patience' epoch\n        if epochs_no_improve >= patience:\n            print(f\"Early stopping at epoch {epoch + 1} due to no improvement.\")\n            break\n\n    # Tải lại mô hình tốt nhất trước khi trả về kết quả\n    model.load_state_dict(torch.load(best_model_path))\n\n    return metrics\n\n\n\n# 4. Định Nghĩa preprocess_files\ndef preprocess_files(file_path, preprocessed_data_dir, desired_shape=(128, 128, 128)):\n    \"\"\"\n    Tiền xử lý một file NIfTI và lưu nó dưới dạng file .npy.\n    \n    Args:\n        file_path (str): Đường dẫn đến file NIfTI.\n        preprocessed_data_dir (str): Đường dẫn đến thư mục lưu trữ các file đã được tiền xử lý.\n        desired_shape (tuple, optional): Kích thước mong muốn của ảnh sau khi thay đổi kích thước.\n    \"\"\"\n    if not os.path.exists(preprocessed_data_dir):\n        os.makedirs(preprocessed_data_dir, exist_ok=True)\n\n    file_name = os.path.basename(file_path)\n    preprocessed_file_path = os.path.join(preprocessed_data_dir, file_name.replace('.nii.gz', '.npy').replace('.nii', '.npy'))\n\n    if os.path.exists(preprocessed_file_path):\n        # File đã được tiền xử lý, bỏ qua\n        return\n\n    try:\n        image = nib.load(file_path).get_fdata()\n        factors = (\n            desired_shape[0] / image.shape[0],\n            desired_shape[1] / image.shape[1],\n            desired_shape[2] / image.shape[2]\n        )\n        resized_image = zoom(image, factors, order=3)\n        resized_image = resized_image.astype(np.float32)\n        np.save(preprocessed_file_path, resized_image)\n    except Exception as e:\n        print(f\"Error processing {file_name}: {e}\")\n\n\n\n# 5. Định Nghĩa hàm main với các cải tiến\ndef main():\n    # Đường dẫn và cài đặt\n    csv_file = '/kaggle/input/unhealthy-csv-file/combined_data (6).csv'\n    raw_data_dir = '/kaggle/input/abdominal-trauma-nii-csv'\n    preprocessed_data_dir = '/kaggle/working/preprocessed_data/'\n    desired_shape = (128, 128, 128)\n    K = 4  # Số lượng folds\n    num_epochs = 200\n    batch_size = 16\n    num_workers = 4\n    num_classes = 14\n    patience = 50\n\n    # Định nghĩa các biến đổi (transform)\n    transform = transforms.Compose([\n        transforms.Normalize(mean=[0.5], std=[0.5])\n    ])\n\n    # Đọc file CSV\n    data = pd.read_csv(csv_file)\n    data.columns = data.columns.str.strip()\n    file_paths = data['file_path'].values\n    labels = data[['bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n                  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low',\n                  'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high', 'any_injury']].values\n\n    # Tiền xử lý dữ liệu\n    print(\"Bắt đầu tiền xử lý dữ liệu...\")\n    for file_path in tqdm(file_paths, desc=\"Preprocessing files\"):\n        full_file_path = os.path.join(raw_data_dir, file_path)\n        if not os.path.exists(full_file_path):\n            print(f\"File không tồn tại: {full_file_path}\")\n            continue\n        preprocess_files(full_file_path, preprocessed_data_dir, desired_shape)\n    print(\"Tiền xử lý dữ liệu hoàn thành.\")\n\n    # Cập nhật đường dẫn file_paths sau khi tiền xử lý\n    preprocessed_paths = [\n        os.path.join(preprocessed_data_dir, os.path.basename(fp).replace('.nii.gz', '.npy').replace('.nii', '.npy'))\n        for fp in file_paths\n    ]\n\n    # Kiểm tra số lượng file đã tiền xử lý\n    preprocessed_paths = [fp for fp in preprocessed_paths if os.path.exists(fp)]\n    labels = labels[:len(preprocessed_paths)]\n\n    # K-fold Cross Validation\n    kfold = KFold(n_splits=K, shuffle=True, random_state=42)\n    fold_metrics = []\n\n    for fold, (train_idx, val_idx) in enumerate(kfold.split(preprocessed_paths)):\n        print(f\"\\n--- Fold {fold + 1} ---\")\n\n        # Tạo các tập huấn luyện và kiểm tra\n        train_paths, val_paths = [preprocessed_paths[i] for i in train_idx], [preprocessed_paths[i] for i in val_idx]\n        train_labels, val_labels = labels[train_idx], labels[val_idx]\n\n        train_dataset = CustomDataset(train_paths, train_labels, transform=transform, desired_shape=desired_shape)\n        val_dataset = CustomDataset(val_paths, val_labels, transform=transform, desired_shape=desired_shape)\n\n        train_loader = DataLoader (train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers)\n        val_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=num_workers)\n\n        # Tạo mô hình và đặt về device\n        model = Simple3DModel(num_classes=num_classes, input_shape=desired_shape).to(device)\n\n        # Định nghĩa tiêu chuẩn mất mát, tối ưu hóa và scheduler\n        criterion = nn.BCEWithLogitsLoss()\n        optimizer = optim.Adam(model.parameters(), lr=0.001)\n        scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=5)\n\n        # Huấn luyện và đánh giá\n        metrics = train_validate_fold(\n            fold, train_loader, val_loader, model, criterion, optimizer, scheduler, device, num_epochs, patience\n        )\n        fold_metrics.append(metrics)\n\n    # Tổng hợp kết quả - Phiên bản cải tiến\n    print(\"\\n--- Tổng kết ---\")\n    print(f\"Tổng hợp kết quả trên {K} folds:\")\n    \n    # Khởi tạo từ điển để lưu trữ kết quả cuối cùng của mỗi metric\n    final_metrics = {\n        'train_loss': [], 'val_loss': [],\n        'train_accuracy': [], 'val_accuracy': [],\n        'sensitivity': [], 'specificity': [], 'f1_score': []\n    }\n\n    # Lấy giá trị cuối cùng của mỗi metric ở mỗi fold\n    for metrics in fold_metrics:\n        for key in final_metrics.keys():\n            values = metrics[key]\n            final_metrics[key].append(values[-1] if len(values) > 0 else 0)\n\n    # In kết quả chi tiết mà không có độ lệch chuẩn\n    print(\"--- Tổng kết ---\")\n    print(f\"Trung bình train_loss: {np.mean(final_metrics['train_loss']):.4f}\")\n    print(f\"Trung bình val_loss: {np.mean(final_metrics['val_loss']):.4f}\")\n    print(f\"Trung bình train_accuracy: {np.mean(final_metrics['train_accuracy']):.4f}\")\n    print(f\"Trung bình val_accuracy: {np.mean(final_metrics['val_accuracy']):.4f}\")\n    print(f\"Trung bình sensitivity: {np.mean(final_metrics['sensitivity']):.4f}\")\n    print(f\"Trung bình specificity: {np.mean(final_metrics['specificity']):.4f}\")\n    print(f\"Trung bình f1_score: {np.mean(final_metrics['f1_score']):.4f}\")\n\n    # Lưu kết quả ra file\n    results_summary = pd.DataFrame(final_metrics)\n    results_summary.to_csv(\"/kaggle/working/final_metrics_summary.csv\", index=False)\n\n    return final_metrics\n\nif __name__ == \"__main__\":\n    # Đặt GPU hoặc CPU\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:32:29.542491Z","iopub.execute_input":"2024-12-01T18:32:29.542898Z","iopub.status.idle":"2024-12-01T22:33:42.429832Z","shell.execute_reply.started":"2024-12-01T18:32:29.542863Z","shell.execute_reply":"2024-12-01T22:33:42.428642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_image_with_seg(volume, volume_seg=[], orientation='Coronal', num_subplots=20):\n    # simply copy\n    if len(volume_seg) == 0:\n        plot_mask = 0\n    else:\n        plot_mask = 1\n        \n    if orientation == 'Coronal':\n        slices = np.linspace(0, volume.shape[2]-1, num_subplots).astype(np.int16)\n        volume = volume.transpose([1, 0, 2])\n        if plot_mask:\n            volume_seg = volume_seg.transpose([1, 0, 2])\n        \n    elif orientation == 'Sagittal':\n        slices = np.linspace(0, volume.shape[2]-1, num_subplots).astype(np.int16)\n        volume = volume.transpose([2, 0, 1])\n        if plot_mask:\n            volume_seg = volume_seg.transpose([2, 0, 1])\n\n    elif orientation == 'Axial':\n        slices = np.linspace(0, volume.shape[0]-1, num_subplots).astype(np.int16)\n           \n    rows = np.max( [np.floor(np.sqrt(num_subplots)).astype(int) - 2, 1])\n    cols = np.ceil(num_subplots/rows).astype(int)\n    \n    fig, ax = plt.subplots(rows, cols, figsize=(cols * 2, rows * 4))\n    fig.tight_layout(h_pad=0.01, w_pad=0)\n    \n    ax = ax.ravel()\n    for this_ax in ax:\n        this_ax.axis('off')\n\n    for counter, this_slice in enumerate( slices ):\n        plt.sca(ax[counter])\n        \n        image = volume[this_slice, :, :]\n        plt.imshow(image, cmap='gray')\n        \n        if plot_mask:\n            mask = np.where(volume_seg[this_slice, :, :], volume_seg[this_slice, :, :], np.nan)\n            plt.imshow(mask, cmap='Set1', alpha=0.5)        \n        \n        \n        \n        \nplot_image_with_seg(volume, volume_seg, orientation='Coronal', num_subplots=10)\nplot_image_with_seg(volume, volume_seg, orientation='Sagittal', num_subplots=10)\nplot_image_with_seg(volume, volume_seg, orientation='Axial', num_subplots=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:31:42.479048Z","iopub.execute_input":"2024-12-01T18:31:42.479326Z","iopub.status.idle":"2024-12-01T18:31:45.775496Z","shell.execute_reply.started":"2024-12-01T18:31:42.479303Z","shell.execute_reply":"2024-12-01T18:31:45.77466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp -r '/kaggle/input/contrails-libraries/pretrainedmodels-0.7.4/' './'\n!cp -r '/kaggle/input/contrails-libraries/efficientnet_pytorch-0.7.1/' './'\n\n!pip -q install /kaggle/input/dicomsdl--0-109-2/dicomsdl-0.109.2-cp310-cp310-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip -q install '/kaggle/input/contrails-libraries/segmentation_models_pytorch-0.3.3-py3-none-any.whl' --no-deps\n!pip -q install /kaggle/input/contrails-model-def1/einops-0.6.1-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-12-01T18:31:45.776842Z","iopub.execute_input":"2024-12-01T18:31:45.777216Z","iopub.status.idle":"2024-12-01T18:32:07.121386Z","shell.execute_reply.started":"2024-12-01T18:31:45.777182Z","shell.execute_reply":"2024-12-01T18:32:07.120235Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('./pretrainedmodels-0.7.4/pretrainedmodels-0.7.4/')\nsys.path.append('./efficientnet_pytorch-0.7.1/efficientnet_pytorch-0.7.1/')\nsys.path.append(\"/kaggle/input/rsna-abd-models-classes/\")\n","metadata":{"execution":{"iopub.status.busy":"2024-12-01T18:32:07.126459Z","iopub.execute_input":"2024-12-01T18:32:07.126771Z","iopub.status.idle":"2024-12-01T18:32:07.131955Z","shell.execute_reply.started":"2024-12-01T18:32:07.126745Z","shell.execute_reply":"2024-12-01T18:32:07.130974Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport copy\nimport time\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport cv2\nfrom PIL import Image\nimport pydicom\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib.pyplot as plt\n\nimport torch\nfrom torch import nn\nimport torch.nn.functional as F\n\nimport timm\nimport segmentation_models_pytorch as smp\nfrom models import *\n\nimport dicomsdl\ndef __dataset__to_numpy_image(self, index=0):\n    info = self.getPixelDataInfo()\n    dtype = info['dtype']\n    if info['SamplesPerPixel'] != 1:\n        raise RuntimeError('SamplesPerPixel != 1')\n    else:\n        shape = [info['Rows'], info['Cols']]\n    outarr = np.empty(shape, dtype=dtype)\n    self.copyFrameData(index, outarr)\n    return outarr\ndicomsdl._dicomsdl.DataSet.to_numpy_image = __dataset__to_numpy_image   \n\n\ntorch.cuda.set_device('cuda:0')","metadata":{"execution":{"iopub.status.busy":"2024-12-01T18:32:07.133125Z","iopub.execute_input":"2024-12-01T18:32:07.133445Z","iopub.status.idle":"2024-12-01T18:32:10.777604Z","shell.execute_reply.started":"2024-12-01T18:32:07.133424Z","shell.execute_reply":"2024-12-01T18:32:10.776898Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install nibabel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:32:10.77875Z","iopub.execute_input":"2024-12-01T18:32:10.779471Z","iopub.status.idle":"2024-12-01T18:32:19.060315Z","shell.execute_reply.started":"2024-12-01T18:32:10.779421Z","shell.execute_reply":"2024-12-01T18:32:19.059073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport pydicom \nimport nibabel as nib\nimport matplotlib.pyplot as plt\nimport seaborn as sn\nimport os\nfrom glob import glob","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:32:19.062059Z","iopub.execute_input":"2024-12-01T18:32:19.062382Z","iopub.status.idle":"2024-12-01T18:32:19.067578Z","shell.execute_reply.started":"2024-12-01T18:32:19.062356Z","shell.execute_reply":"2024-12-01T18:32:19.066632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install einops\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:32:21.206942Z","iopub.execute_input":"2024-12-01T18:32:21.207223Z","iopub.status.idle":"2024-12-01T18:32:29.540475Z","shell.execute_reply.started":"2024-12-01T18:32:21.207201Z","shell.execute_reply":"2024-12-01T18:32:29.53935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patient_id = df[df['series_id'] == id]['patient_id'].iloc[0]\npatient_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:34:07.089897Z","iopub.execute_input":"2024-12-01T22:34:07.090144Z","iopub.status.idle":"2024-12-01T22:34:07.096749Z","shell.execute_reply.started":"2024-12-01T22:34:07.090123Z","shell.execute_reply":"2024-12-01T22:34:07.095887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seg_filepath = f'{PATH}/segmentations/{id}.nii'\nimg_filepath = f'{PATH}/train_images/{patient_id}/{id}'\n\nprint(\"Segmentation File Path:\", seg_filepath)\nprint(\"Image File Path:\", img_filepath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:34:07.097844Z","iopub.execute_input":"2024-12-01T22:34:07.098153Z","iopub.status.idle":"2024-12-01T22:34:07.106115Z","shell.execute_reply.started":"2024-12-01T22:34:07.098122Z","shell.execute_reply":"2024-12-01T22:34:07.105277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_3D_scans(folder, downsample_rate=1):\n    \n    #reads the filenames in the folder, extracts the IDs from the filenames, sorts them, and constructs a list of filenames to read in order.\n    filenames = os.listdir(folder)\n    filenames = [int(filename.split('.')[0]) for filename in filenames]\n    filenames = sorted(filenames)\n    filenames = [str(filename) + '.dcm' for filename in filenames]\n        \n    volume = []\n    for filename in filenames[::downsample_rate]:\n        filepath = os.path.join(folder, filename)\n        ds = pydicom.dcmread(filepath)\n        #This extracts the pixel array data from the DICOM object. The pixel array represents the 2D image slice\n        image = ds.pixel_array\n        \n        # find rescale params\n        if (\"RescaleIntercept\" in ds) and (\"RescaleSlope\" in ds):\n            intercept = float(ds.RescaleIntercept)\n            slope = float(ds.RescaleSlope)\n    \n        # find clipping params\n        center = int(ds.WindowCenter)\n        width = int(ds.WindowWidth)\n        low = center - width / 2\n        high = center + width / 2    \n        image = (image * slope) + intercept\n        image = np.clip(image, low, high)\n\n        image = (image / np.max(image) * 255).astype(np.int16)#Normalize the pixel values to a range between 0 and 255 and convert them to 16-bit integers. This step is usually done for visualization purpose\n        image = image[::downsample_rate, ::downsample_rate]\n        volume.append( image )\n    \n    volume = np.stack(volume, axis=0)\n    return volume","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:34:07.107245Z","iopub.execute_input":"2024-12-01T22:34:07.107496Z","iopub.status.idle":"2024-12-01T22:34:07.115668Z","shell.execute_reply.started":"2024-12-01T22:34:07.107465Z","shell.execute_reply":"2024-12-01T22:34:07.114827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Set the model to evaluation mode\nmodel.eval()\n\n# Select a random image from the validation dataset\nrandom_index = np.random.randint(len(dataset_val))\nimage, label = dataset_val[random_index]\n\n# Move the image to the GPU if available\nimage = image.to('cuda')\n\n# Pass the image through the model\nwith torch.no_grad():\n    output = model(image.unsqueeze(0))  # Unsqueeze to add batch dimension\n\n# Convert the output logits to probabilities using sigmoid function\npredicted_probs = torch.sigmoid(output)[0]\n\n# Convert predicted probabilities to binary predictions\npredicted_labels = (predicted_probs > 0.5).int()\n\n\n# Display the image, actual labels, and predicted labels\nplt.imshow(image.permute(1, 2, 0).cpu())  # Move image to CPU and change channel order\n#plt.title(f\"Actual Labels: {label}\\nPredicted Labels: {predicted_labels}\")\nplt.title(f\"Actual Labels: {label}\\nPredicted Labels: {predicted_labels}\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocess Util","metadata":{}},{"cell_type":"code","source":"def glob_sorted(path):\n    return sorted(glob(path), key=lambda x: int(x.split('/')[-1].split('.')[0]))\n\ndef get_rescaled_image(dcm, img):\n    resI, resS = dcm.RescaleIntercept, dcm.RescaleSlope\n    img = resS * img + resI\n    return img\n\ndef get_windowed_image(img, WL=50, WW=400):\n    upper, lower = WL+WW//2, WL-WW//2\n    X = np.clip(img.copy(), lower, upper)\n    X = X - np.min(X)\n    X = X / np.max(X)\n    X = (X*255.0).astype('uint8')\n    \n    return X\n\ndef standardize_pixel_array(dcm, pixel_array):\n    \"\"\"\n    Source : https://www.kaggle.com/competitions/rsna-2023-abdominal-trauma-detection/discussion/427217\n    \"\"\"\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    #pixel_array = dcm.pixel_array\n    \n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        pixel_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n\n    intercept = float(dcm.RescaleIntercept)\n    slope = float(dcm.RescaleSlope)\n    center = int(dcm.WindowCenter)\n    width = int(dcm.WindowWidth)\n    low = center - width / 2\n    high = center + width / 2    \n    \n    pixel_array = (pixel_array * slope) + intercept\n    pixel_array = np.clip(pixel_array, low, high)\n\n    return pixel_array\n\ndef load_volume(dcms):\n    volume = []\n    pos_zs = []\n    \n    for dcm_path in dcms:\n        pydcm = pydicom.dcmread(dcm_path)\n        \n        pos_z = pydcm[(0x20, 0x32)].value[-1]\n        pos_zs.append(pos_z)\n        \n        dcm = dicomsdl.open(dcm_path)\n        \n        orig_image = dcm.to_numpy_image()\n        image = get_rescaled_image(dcm, orig_image)\n        image = get_windowed_image(image)\n        \n        if np.min(image)<0:\n            image = image + np.abs(np.min(image))\n        \n        image = image / image.max()\n        image = (image * 255).astype(np.uint8)\n        volume.append(image)\n    \n    return np.stack(volume)\n\n\ndef process_volume(volume):\n    volume = np.stack([cv2.resize(x, (128, 128)) for x in volume])\n    \n    volumes = []\n    cuts = [(x, x+32) for x in np.arange(0, volume.shape[0], 32)[:-1]]\n    \n    if cuts:\n        for cut in cuts:\n            volumes.append(volume[cut[0]:cut[1]])\n        volumes = np.stack(volumes)\n    else:\n        volumes = np.zeros((1, 32, 128, 128), dtype=np.uint8)\n        volumes[0, :len(volume)] = volume\n    \n    if cuts:\n        last_volume = np.zeros((1, 32, 128, 128), dtype=np.uint8)\n        last_volume[0, :volume[cuts[-1][1]:].shape[0]] =  volume[cuts[-1][1]:]\n        volumes = np.concatenate([volumes, last_volume])\n    \n    volumes = torch.as_tensor(volumes).float()\n    \n    return volumes\n\n\ndef get_volume_data(grd, step=96, stride=1, stride_cutoff=200):\n    volumes = []\n    \n    if len(grd)>stride_cutoff:\n        grd = grd[::stride]\n\n    take_last = False\n    if not str(len(grd)/step).endswith('.0'):\n        take_last = True\n\n    started = False\n    for i in range(len(grd)//step):\n        rows = grd[i*step:(i+1)*step]\n\n        if len(rows)!=step:\n            rows = pd.DataFrame([rows.iloc[int(x*len(rows))] for x in np.arange(0, 1, 1/step)])\n\n        volumes.append(rows)\n\n        started = True\n\n    if not started:\n        rows = grd\n        rows = pd.DataFrame([rows.iloc[int(x*len(rows))] for x in np.arange(0, 1, 1/step)])\n        volumes.append(rows)\n\n    if take_last:\n        rows = grd[-step:]\n        if len(rows)==step:\n            volumes.append(rows)\n\n    return volumes","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:34:41.182531Z","iopub.execute_input":"2024-12-01T22:34:41.182866Z","iopub.status.idle":"2024-12-01T22:34:41.200536Z","shell.execute_reply.started":"2024-12-01T22:34:41.182834Z","shell.execute_reply":"2024-12-01T22:34:41.199595Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"IMAGE_FOLDER = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/'\n\npatient = '10004'  # We only predict a single patient in this notebook\n\ntest_augs = A.Compose([\n    A.Resize(384, 384),\n    ToTensorV2()\n])\n\npatient","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:34:41.204801Z","iopub.execute_input":"2024-12-01T22:34:41.205081Z","iopub.status.idle":"2024-12-01T22:34:41.215387Z","shell.execute_reply.started":"2024-12-01T22:34:41.20506Z","shell.execute_reply":"2024-12-01T22:34:41.214289Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"path = f'/kaggle/input/rsna-abd-models/try3_seg_resnet18d_v3/zip/0.pth'\nst = torch.load(path, map_location='cpu')\nmodel_3dseg = convert_3d(SegmentationModel())\nmodel_3dseg.load_state_dict(st)\nmodel_3dseg.eval()\nmodel_3dseg.cuda()\n    \n    \npath = f\"/kaggle/input/coatmed384ourdataseed6969/3.pth\"\nst = torch.load(path, map_location='cpu')\nmodel_organs = Model4(num_classes=10, seg_classes=4, arch='medium', mask_head=False)\nmodel_organs.load_state_dict(st)\nmodel_organs.cuda()\nmodel_organs.eval()\n\n\npath = f\"/kaggle/input/coatsmall384extravast4funet/3.pth\"\nst = torch.load(path, map_location='cpu')\nmodel_extrav = Model4(num_classes=2, seg_classes=4, arch='small', mask_head=False)\nmodel_extrav.load_state_dict(st)\nmodel_extrav.cuda()\nmodel_extrav.eval()\n\nprint('''We load 3 models here:\n1: 3d semantic segmentation model for segment organs\n2: 2.5d classification model for classify organs\n3: 2.5d classification model for classify extravasation\n''')","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:34:41.216497Z","iopub.execute_input":"2024-12-01T22:34:41.216842Z","iopub.status.idle":"2024-12-01T22:34:46.662704Z","shell.execute_reply.started":"2024-12-01T22:34:41.216811Z","shell.execute_reply":"2024-12-01T22:34:46.661733Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"PATIENT_TO_PREDICTION = {}\nPATIENT_TO_PREDICTION2 = {}\n\nfinal_outputs = []\nfinal_outputs2 = []\n\nstudies = os.listdir(f'{IMAGE_FOLDER}/{patient}')\nfor study in studies:\n\n    files = glob_sorted(f\"{IMAGE_FOLDER}/{patient}/{study}/*\")\n\n    volume = load_volume(files)\n    file_to_volume = {file: vol for file, vol in zip(files, volume)}\n\n    volumes = process_volume(volume)\n    volumes_seg = predict_segmentation(volumes, [model_3dseg])\n    volume_seg = np.concatenate(volumes_seg.transpose(0, 2, 1, 3, 4))[:len(volume)]\n    \n    vis_seg_0 = volumes[volumes.shape[0]//2, 16].numpy().astype(np.uint8)\n    vis_seg_1 = volumes_seg[volumes_seg.shape[0]//2, :, 16]\n    vis_seg_1[0] += vis_seg_1[3]\n    vis_seg_1[1] += vis_seg_1[4]\n    vis_seg_1 = (vis_seg_1[:3].transpose(1,2,0).clip(0, 1) * 255).astype(np.uint8)\n\n#     print(volumes.shape, volumes_seg.shape)  # torch.Size([7, 32, 128, 128]) (7, 5, 32, 128, 128)\n\n    msk = volume_seg.max(0).max(0)\n    ys, xs = np.where(msk)\n    y1, y2, x1, x2 = np.min(ys) / 128, np.max(ys) / 128, np.min(xs) / 128, np.max(xs) / 128\n\n    files = pd.DataFrame({\"file\": files})\n    files_volumes = get_volume_data(files, step=96, stride=2, stride_cutoff=400)\n\n    first = True\n\n    del volumes, volumes_seg, volume_seg, volume\n    gc.collect()\n\n    for file_volume in files_volumes:\n        volume = np.stack([file_to_volume[file] for file in file_volume.file])\n\n        if first:\n            h, w = volume.shape[1:]\n            y1, y2, x1, x2 = int(y1*h), int(y2*h), int(x1*w), int(x2*w)\n        volume2 = volume\n        \n        #### CROPPED #####\n        volume = volume[:, y1:y2, x1:x2]\n\n        vols = []\n        NC = 3\n        for i in range(len(volume)//NC):\n            vols.append(volume[i*NC:(i+1)*NC])\n        vol = np.stack(vols, 0).transpose(0, 2, 3, 1)\n\n        volume_ = []\n        for image in vol:\n            image = image.astype(np.float32) / 255\n            transformed = test_augs(image=image)\n            image = transformed['image']\n            volume_.append(image)\n        volume = torch.stack(volume_).float()\n        volume = volume.cuda()\n\n        #### UNCROPPED #####\n        vols = []\n        NC = 3\n        for i in range(len(volume2)//NC):\n            vols.append(volume2[i*NC:(i+1)*NC])\n        vol = np.stack(vols, 0).transpose(0, 2, 3, 1)\n\n        volume_ = []\n        for image in vol:\n            image = image.astype(np.float32) / 255\n            transformed = test_augs(image=image)\n            image = transformed['image']\n            volume_.append(image)\n\n        volume2 = torch.stack(volume_).float()\n        volume2 = volume2.cuda()\n\n        outputs = []\n        outputs2 = []\n\n        with torch.no_grad():\n            with torch.cuda.amp.autocast(enabled=True):\n\n                outs = model_organs(volume.unsqueeze(0))\n                outs = outs.float().sigmoid()\n                outputs.append(outs)\n\n                outs = model_extrav(volume2.unsqueeze(0))[:, :, [1, 0]]\n                outs = outs.float().sigmoid()\n                outputs2.append(outs)\n\n        torch.cuda.empty_cache()\n\n        outputs = torch.stack(outputs)[:, 0].mean(0)\n        outputs2 = torch.stack(outputs2)[:, 0].mean(0)\n\n        final_outputs.append(outputs.detach().cpu().numpy())\n        final_outputs2.append(outputs2.detach().cpu().numpy())\n\n        first = False\n\n        torch.cuda.empty_cache()\n\n\nlast_final_outputs = final_outputs.copy()\nlast_final_outputs2 = final_outputs2.copy()\n\nfinal_outputs = np.concatenate(final_outputs)\nfinal_outputs2 = np.concatenate(final_outputs2)\n\nfinal_predictions = final_outputs.max(0)\nfinal_predictions2 = final_outputs2.max(0)\n\nPATIENT_TO_PREDICTION[patient] = final_predictions\nPATIENT_TO_PREDICTION2[patient] = final_predictions2\n","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:34:46.663639Z","iopub.execute_input":"2024-12-01T22:34:46.663893Z","iopub.status.idle":"2024-12-01T22:35:23.216976Z","shell.execute_reply.started":"2024-12-01T22:34:46.663871Z","shell.execute_reply":"2024-12-01T22:35:23.21623Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axarr = plt.subplots(1, 2, figsize=(10, 4))\n\naxarr[0].imshow(vis_seg_0)\naxarr[0].axis('off') \n\naxarr[1].imshow(vis_seg_1)\naxarr[1].axis('off') \n\nplt.tight_layout() \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:35:23.217943Z","iopub.execute_input":"2024-12-01T22:35:23.218207Z","iopub.status.idle":"2024-12-01T22:35:23.468066Z","shell.execute_reply.started":"2024-12-01T22:35:23.218184Z","shell.execute_reply":"2024-12-01T22:35:23.467214Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Postprocess","metadata":{}},{"cell_type":"code","source":"bowel_w = 2\nextrav_w = 6\nlow_w = 2\nhigh_w = 4\n\nFINAL_SUB = {'patient_id': [], 'bowel_healthy': [], 'bowel_injury': [], \n             'extravasation_healthy': [], 'extravasation_injury': [], \n             'kidney_healthy': [], 'kidney_low': [], 'kidney_high': [],\n             'liver_healthy': [], 'liver_low': [], 'liver_high': [],\n             'spleen_healthy': [], 'spleen_low': [], 'spleen_high': [],}\n\nfor patient in PATIENT_TO_PREDICTION:\n    prediction = PATIENT_TO_PREDICTION[patient].copy()\n    prediction2 = PATIENT_TO_PREDICTION2[patient].copy()\n    \n    prediction[9] = (prediction[9] * 0.666) + (prediction2[1]*0.334)\n    \n    FINAL_SUB['patient_id'].append(patient)\n    \n    FINAL_SUB['bowel_healthy'].append(1 - prediction[9])\n    FINAL_SUB['bowel_injury'].append(prediction[9] * bowel_w)\n    FINAL_SUB['extravasation_healthy'].append(1 - prediction2[0])\n    FINAL_SUB['extravasation_injury'].append(0.06355258976803305 + (prediction2[0] * extrav_w))\n    \n    FINAL_SUB['liver_healthy'].append(1 - prediction[0])\n    FINAL_SUB['liver_low'].append(prediction[3]*low_w)\n    FINAL_SUB['liver_high'].append(prediction[4]*high_w)\n\n    FINAL_SUB['spleen_healthy'].append(1 - prediction[1])\n    FINAL_SUB['spleen_low'].append(prediction[5]*low_w)\n    FINAL_SUB['spleen_high'].append(prediction[6]*high_w)\n\n    FINAL_SUB['kidney_healthy'].append(1 - prediction[2])\n    FINAL_SUB['kidney_low'].append((prediction[7])*low_w)\n    FINAL_SUB['kidney_high'].append(prediction[8]*high_w)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:35:23.4692Z","iopub.execute_input":"2024-12-01T22:35:23.469467Z","iopub.status.idle":"2024-12-01T22:35:23.477803Z","shell.execute_reply.started":"2024-12-01T22:35:23.469445Z","shell.execute_reply":"2024-12-01T22:35:23.476724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nsubmission = pd.DataFrame(FINAL_SUB)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:35:23.478678Z","iopub.execute_input":"2024-12-01T22:35:23.478884Z","iopub.status.idle":"2024-12-01T22:35:23.500192Z","shell.execute_reply.started":"2024-12-01T22:35:23.478867Z","shell.execute_reply":"2024-12-01T22:35:23.499205Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patient_id = submission['patient_id'].iloc[0]\ndata_to_plot = submission.drop(columns=['patient_id'])\nenglish_column_names = {\n    \"bowel_healthy\": \"Bowel Healthy\",\n    \"bowel_injury\": \"Bowel Injury\",\n    \"extravasation_healthy\": \"Extravasation Healthy\",\n    \"extravasation_injury\": \"Extravasation Injury\",\n    \"kidney_healthy\": \"Kidney Healthy\",\n    \"kidney_low\": \"Kidney Low\",\n    \"kidney_high\": \"Kidney High\",\n    \"liver_healthy\": \"Liver Healthy\",\n    \"liver_low\": \"Liver Low\",\n    \"liver_high\": \"Liver High\",\n    \"spleen_healthy\": \"Spleen Healthy\",\n    \"spleen_low\": \"Spleen Low\",\n    \"spleen_high\": \"Spleen High\"\n}\n\ndata_to_plot.columns = [english_column_names[col] for col in data_to_plot.columns]\n\nplt.figure(figsize=(15, 7))\ndata_to_plot.T.plot(kind='bar', legend=False, ax=plt.gca())\nplt.title(f\"Probability of Diseases for Patient {patient_id}\")\nplt.ylabel(\"Probability\")\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:35:23.501479Z","iopub.execute_input":"2024-12-01T22:35:23.501921Z","iopub.status.idle":"2024-12-01T22:35:23.907758Z","shell.execute_reply.started":"2024-12-01T22:35:23.501887Z","shell.execute_reply":"2024-12-01T22:35:23.906913Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -rf /kaggle/working/*","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:35:23.908865Z","iopub.execute_input":"2024-12-01T22:35:23.9092Z","iopub.status.idle":"2024-12-01T22:35:26.611098Z","shell.execute_reply.started":"2024-12-01T22:35:23.909151Z","shell.execute_reply":"2024-12-01T22:35:26.609811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-01T22:35:26.612762Z","iopub.execute_input":"2024-12-01T22:35:26.613051Z","iopub.status.idle":"2024-12-01T22:35:26.62043Z","shell.execute_reply.started":"2024-12-01T22:35:26.613028Z","shell.execute_reply":"2024-12-01T22:35:26.619625Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}