{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52254,"databundleVersionId":6863140,"sourceType":"competition"}],"dockerImageVersionId":30528,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n\nimport numpy as np\nimport pandas as pd \n\n\n#This part of the code is a simple loop that iterates through the directory structure under '/kaggle/input' and prints the paths of the first 10 files encountered. \n#It uses the os.walk() function to traverse the directory tree.\n\nimport os\ni=0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        i+=1\n        print(os.path.join(dirname, filename))\n        if i==10:\n            break\n    if i==10:\n        break\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pydicom","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing necessary libraries\n'''\nThis section imports the necessary libraries and modules for working with PyTorch, \nimage processing, data loading, and visualization.\n'''\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\nimport torchvision.transforms as transforms\nimport torch\nimport torchvision.models as models\nimport torchvision.models as models\nfrom PIL import Image\nimport shutil\nimport matplotlib.pyplot as plt\nimport pydicom\nfrom glob import glob\nimport random","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nHere, pre-trained ResNet-18 and DenseNet121 models are loaded, \nmodified to accept grayscale images, and then their state dictionaries are saved for future use.\npython\n'''\n\n# Save ResNet-18\nresnet_model = models.resnet18(pretrained=True)\nresnet_model.conv1 = nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\ntorch.save(resnet_model.state_dict(), '/kaggle/working/resnet18.pth')\n\n# Save DenseNet121\ndensenet_model = models.densenet121(pretrained=True)\ndensenet_model.features.conv0 = nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\ntorch.save(densenet_model.state_dict(), '/kaggle/working/densenet121.pth')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\ni=0\nfor dirname, _, filenames in os.walk('/kaggle/working/'):\n    for filename in filenames:\n        i+=1\n        print(os.path.join(dirname, filename))\n        if i==10:\n            break\n    if i==10:\n        break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis section defines hyperparameters like batch size, learning rate, \nand the number of epochs for training.\n'''\n\n# Define hyperparameters\nbatch_size = 32\nlearning_rate = 0.001\nnum_epochs = 3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nA data transformation pipeline is defined using torchvision.transforms.Compose. \nIt includes resizing images to 224x224,converting them to tensors, and normalizing pixel values.\n'''\n\n#Define data transformation\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    #transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    transforms.Normalize(mean=[0.5], std=[0.5])\n])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis line reads a CSV file containing labels for each patient. \nThe labels represent the presence or absence of injuries in various organs.\n'''\n\n# Load the CSV file with labels\nlabels_df = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n'''\nThis section creates a dictionary (labels_dict) to store the labels for each patient based on the information in the CSV file.\n'''\n\n# Create a dictionary to store labels for each patient\nlabels_dict = {row['patient_id']: row[['bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n                                       'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low',\n                                       'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high']].values\n               for _, row in labels_df.iterrows()}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nA custom dataset class is defined for loading DICOM images along with their labels. \nIt includes methods like __len__ and __getitem__ for handling dataset operations.\n'''\n\n\n# Defining custom dataset\nclass CustomDataset(torch.utils.data.Dataset):\n    def __init__(self, data_dir, labels_dict, transform=None):\n        self.data_dir = data_dir\n        self.labels_dict = labels_dict\n        self.transform = transform\n        self.image_paths = glob(os.path.join(data_dir, '**', '*.dcm'), recursive=True)\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, index):\n        #print(f'Loading item {index}')\n        image_path = self.image_paths[index]\n        dicom_data = pydicom.dcmread(image_path)\n        pixel_data = dicom_data.pixel_array\n        \n        # Normalize pixel data to range [0, 1]\n        pixel_data = pixel_data / 255.0  # pixel values are in the range [0, 255]\n        \n        # Convert pixel data to a PIL image\n        image = Image.fromarray(pixel_data)\n        \n        \n        patient_id = int(image_path.split('/')[-3])  # Extract patient ID from path\n        label = torch.tensor(self.labels_dict[patient_id], dtype=torch.float32).float()\n\n        if self.transform is not None:\n            image = self.transform(image)\n        #print(f'Patient ID: {patient_id}')\n        return image, label","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''The custom dataset is instantiated and used to create a DataLoader for training.'''\n\n\n#Loading Train Dataset and Trainloader\n\ntrain_dataset = CustomDataset(data_dir, labels_dict, transform=transform)\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This section chooses and displays an image from the dataset to visualize and decide the similarities.'''\n\n# Choosing and Displaying an image to decide similarities\n# This image choosed from the middle part of the body where the organ we interested.\n# This images will use to find most similar images for each patient.\n# It could be change depends on the patient for future studies.\n\nfrom collections import defaultdict\npatient_images = defaultdict(list)\nfor image_path in train_dataset.image_paths:\n    patient_id = int(image_path.split('/')[-3])\n    patient_images[patient_id].append(image_path)\n    \ndicom_data = pydicom.dcmread(patient_images[23424][0])\npixel_data = dicom_data.pixel_array / 255.0  # Normalize pixel data\nimage = Image.fromarray(pixel_data)\nif transform is not None:\n    image = transform(image)\nplt.imshow(image.squeeze(), cmap='gray')\nplt.show();","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''The code snippet calculates and displays the number of patients in the patient_images dictionary and \nthe count of unique patient IDs in the labels_df DataFrame, \nproviding insights into the dataset's size and patient ID uniqueness.'''\n\n# patient_images is a dictionary that includes the images for patient by batient \n\nlen(patient_images) , labels_df.patient_id.nunique() # shows how many unic patient we have","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This line of code calculates and displays the count of patients in the dataset who have at least one injury, \nusing the 'any_injury' column from the labels_df DataFrame.'''\n\n# showing the number of patient who has at least one injury\n\nsum(labels_df.any_injury)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This code snippet identifies patients with at least one injury by filtering the labels_df DataFrame based on the 'any_injury' column. \nIt then creates a list (injury_patientid) containing the patient IDs of those with at least one injury. \nThe length of this list is subsequently calculated, representing the count of patients with at least one injury in the dataset.'''\n\n# showing the number of patient who has at least one injury\n\ninjury_patientid=list(labels_df.query('any_injury == 1')['patient_id'])\nlen(injury_patientid)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This code snippet prints the first 10 patient IDs from the list injury_patientid, which contains the patient IDs of individuals with at least one recorded injury. \nIt provides a quick check to inspect and verify the results of the filtering operation.'''\n\n# checking the result\n\ninjury_patientid[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''This code creates a new dictionary named injury_patient_images by selectively including only the entries from the patient_images dictionary where the patient ID corresponds to individuals with at least one recorded injury. \nThe function select_items_by_keys facilitates this selection based on the list of patient IDs (injury_patientid). \nThe printed result shows the first 10 patient IDs from the newly created dictionary, providing a glimpse of patients with injuries and their associated images.'''\n\n# Create a dictionary which includes patients that have at least one injury from patient images dictionary\n\ndef select_items_by_keys(input_dict, keys_list):\n    return {key: input_dict[key] for key in keys_list if key in input_dict}\n\n# Example usage \n# patient_images is a dictionary that has all images for each patient\n# injury_patientid is a list that has patient id who has any injury\n\n\ninjury_patient_images = select_items_by_keys(patient_images, injury_patientid)\nprint(list(injury_patient_images.keys())[:10])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis line of code creates a new dictionary named `reduced_patient_images` by making a copy of the existing `patient_images` dictionary. This copy is created to preserve the original dataset while allowing modifications or operations specific to a reduced set of patient images in subsequent code cells.\n'''\n# copying the images to a new dictionary. This dictionary will be used in the next code cells.\nreduced_patient_images = patient_images.copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"'''\nThis code snippet checks the length of the reduced_patient_images dictionary, providing a quick verification to ensure that the copying process was successful and that the new dictionary contains the expected number of patient entries.\n\n\n\n\n\n\n\n'''\n# checking the result that everythig is fine.\nlen(reduced_patient_images)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n this code uses the delete_random_items function to randomly remove patient IDs with no injury from the reduced_patient_images dictionary. \n This removal is intended to balance the dataset by reducing the number of patients without losing those who have at least one recorded injury. \n The final print statement is used to confirm the number of items in the reduced_patient_images dictionary after the deletion process.\n'''\n\n\n# Our dataset is an imbalanced dataset. This dataset should be made as a balanced dataset without don't lose the patient who has at least one injury.\n# We have total 3147 patient id. 855 of them has any injury. The dataset will be decreased the number of patient id from 3147 to 1710. \n# 855 of 1710 has any injury. 855 of 1710 has no injury. These no injury patient ids will be choosed as randomly. \n# To remove 1437 patient who has no injury will increase the balancity and also the accuracy of the prediction of model. \n\ndef delete_random_items(target_dict, reference_dict, num_items_to_delete):\n    target_keys = list(target_dict.keys())\n    reference_keys = set(reference_dict.keys())\n\n    common_keys = set(target_keys) & reference_keys\n    exclusive_keys = set(target_keys) - common_keys\n\n    keys_to_delete = random.sample(exclusive_keys, num_items_to_delete)\n\n    for key in keys_to_delete:\n        target_dict.pop(key)\n\n\nnum_items_to_delete = 1437\n\ndelete_random_items(reduced_patient_images, injury_patient_images, num_items_to_delete)\n\nprint(len(reduced_patient_images))  # To verify the number of items in target_dict after deletion","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code snippet loads a pre-trained ResNet-18 model and modifies it for finding similar medical images. Specifically:\n\nThe model is configured to accept single-channel (grayscale) images instead of three-channel (RGB) images.\nPre-trained weights for the modified ResNet model are loaded.\nThe last classification layer is removed from the model, leaving it prepared for similarity comparison of medical images.\n'''\n\n\n# We don't need all images for a patient. A patient has many unnecessary images. We already have an image from the middle part of the body. \n# It is easy to find most similar images to our referance images thanks to our pre-trained model RESNET\n\n# Load pre-trained ResNet model to find similarities and get rid of unnecessary images\nmodel = models.resnet18(pretrained=False)\n\n# Modify the first layer to accept 1 channel instead of 3\nmodel.conv1 = nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\nmodel.load_state_dict(torch.load('/kaggle/working/resnet18.pth'))\n\n# Remove the last classification layer\nmodel = nn.Sequential(*list(model.children())[:-1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n\nThis code section loads and preprocesses a reference medical image from a specific patient (ID: 23424). It normalizes the pixel data, converts it to a PIL image, and applies an optional transformation. \nThe resulting image is then formatted as a tensor with batch and channel dimensions, preparing it as a query image for similarity comparison.\n'''\n\n# cheking the reference image\n# Load and preprocess the query image\n\ndicom_data = pydicom.dcmread(patient_images[23424][0])\npixel_data = dicom_data.pixel_array / 255.0  # Normalize pixel data\nimage = Image.fromarray(pixel_data)\nif transform is not None:\n    image = transform(image)\n\nquery_image = image.unsqueeze(0)# Add batch and channel dimensions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code extracts features from the pre-processed query image using the modified ResNet model. \nThe resulting features are stored in a one-dimensional tensor, representing a condensed representation of the image for similarity comparison.\n'''\n\n# Extracting features\nwith torch.no_grad():\n    features = model(query_image)\n\nfeatures = features.squeeze()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code calculates and stores the top 50 similar images for each patient in a dictionary (`top_similar_images`). \nIt uses a pre-trained ResNet model to extract features and calculates cosine similarity between a reference image and each patient's images. \nThe resulting dictionary provides ranked lists of similar images for each patient based on their similarity scores.\n'''\n\n#SIMILARITIES FOR PATIENT BY PATIENT AS A DICTIONARY FORMAT (IN 100 IMAGES TOP 50 IMAGES)\n\n# Create a new dictionary to store top similar images for each patient\ntop_similar_images = defaultdict(list)\nmax_image_in_patient = 100\ntst=0\n# Iterate through each patient's images\nfor patient_id, images in reduced_patient_images.items():\n    # Iterate through each image for the patient\n    for image_path in images:\n        dicom_data = pydicom.dcmread(image_path)\n        pixel_data = dicom_data.pixel_array / 255.0\n        image = Image.fromarray(pixel_data)\n        if transform is not None:\n            image = transform(image)\n        image = image.unsqueeze(0)\n\n        with torch.no_grad():\n            image_features = model(image)\n        \n        # Calculate similarity with the selected image\n        similarity = np.dot(features.squeeze(), image_features.squeeze())  # Cosine similarity\n        \n        # Add the image path and similarity score to the list for this patient\n        top_similar_images[patient_id].append((image_path, similarity))\n        \n        if len(top_similar_images[patient_id])>=max_image_in_patient:\n            tst+=1\n            break\n        \n    if tst==2:\n        break\n# Sort the images by similarity score for each patient\nfor patient_id, images in top_similar_images.items():\n    images.sort(key=lambda x: x[1], reverse=True)\n    top_similar_images[patient_id] = images[:50]  # Keep only the top 100 similar images","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code snippet returns the number of patients for which similar images have been calculated and stored in the `top_similar_images` dictionary.\n'''\nlen(top_similar_images)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code snippet visualizes a specific image from the top similar images. \nIt loads the DICOM data, normalizes pixel values, converts it to a PIL image, and displays the image using matplotlib. \nThe displayed image is from the set of top similar images for a particular patient.\n'''\n\n\ndicom_data = pydicom.dcmread(top_similar_images[32627][8][0])\npixel_data = dicom_data.pixel_array / 255.0  # Normalize pixel data\nimage = Image.fromarray(pixel_data)\nif transform is not None:\n    image = transform(image)\nplt.imshow(image.squeeze(), cmap='gray')\nplt.show();","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code defines a custom dataset (`TopSimilarImagesDataset`) and creates a corresponding data loader (`top_similar_images_loader`). \nThe dataset is designed for training a model on top similar medical images, and the data loader facilitates efficient batch processing during model training.\n'''\n\n\n# CREATING TNE NEW DATA TRAIN LOADER FOR THE MODEL THAT PREDICT THE LABELS\n\n\n# Define a new custom dataset for the top similar images\nclass TopSimilarImagesDataset(torch.utils.data.Dataset):\n    def __init__(self, similar_images, labels_dict, transform=None):\n        self.similar_images = similar_images\n        self.labels_dict = labels_dict\n        self.transform = transform\n\n    def __len__(self):\n        return sum(len(images) for images in self.similar_images.values())\n\n    def __getitem__(self, index):\n        # Find the corresponding patient and image index\n        for patient_id, images in self.similar_images.items():\n            if index < len(images):\n                image_path, _ = images[index]\n                break\n            index -= len(images)\n        \n        dicom_data = pydicom.dcmread(image_path)\n        pixel_data = dicom_data.pixel_array / 255.0\n        image = Image.fromarray(pixel_data)\n        if self.transform is not None:\n            image = self.transform(image)\n        \n        patient_id = int(image_path.split('/')[-3])\n        label = torch.tensor(self.labels_dict[patient_id], dtype=torch.float32).float()\n\n        return image, label\n\n# Create an instance of the new dataset\ntop_similar_images_dataset = TopSimilarImagesDataset(top_similar_images, labels_dict, transform=transform)\n\n# Create a DataLoader for the new dataset\ntop_similar_images_loader = DataLoader(top_similar_images_dataset, batch_size=batch_size, shuffle=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nThis code defines a custom neural network model (`CustomModel`) for predicting labels based on medical images. \nThe model uses a modified DenseNet121 architecture, with the first layer adjusted to accept a single channel. It is designed for multi-label classification with 13 output classes. Pre-trained weights for DenseNet121 are loaded from a specified path. \nThe model is initialized and ready for training on the provided medical image dataset.\n'''\n\n# CREATING MODEL TO PRODUCE LABELS\n\nclass CustomModel(nn.Module):\n    def __init__(self, num_classes):\n        super(CustomModel, self).__init__()\n        self.base_model = models.densenet121(pretrained=False)\n        in_features = self.base_model.classifier.in_features\n        \n        # Modify the first layer to accept 1 channel instead of 3\n        self.base_model.features.conv0 = nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n        densenet_model.load_state_dict(torch.load('/kaggle/working/densenet121.pth'))\n\n        \n        self.base_model.classifier = nn.Sequential(\n            nn.Linear(in_features, num_classes),\n            #nn.Sigmoid()  # Sigmoid for multi-label classification\n            nn.Softmax(dim=1)  # Softmax for multi-label classification\n        )\n        \n    def forward(self, x):\n        return self.base_model(x)\n\nmodel = CustomModel(num_classes=13)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This section of code involves the training and testing of a neural network model for predicting labels on medical images. Here's a brief breakdown:\n\n1. **Defining Loss and Optimization:**\n   - `criterion` is defined as `nn.BCEWithLogitsLoss()`, indicating Binary Cross Entropy Loss with Logits.\n   - `optimizer` is set using Adam with a specified learning rate.\n\n2. **Training the Model:**\n   - The model is trained using a loop over epochs and batches.\n   - The specified print_every variable determines how often the loss is printed during training.\n   - The training loop prints information about the epoch, batch, and loss.\n\n3. **Saving the Model:**\n   - The trained model is saved with its state dictionary using `torch.save()`.\n\n4. **Testing the Model on New Images:**\n   - The saved model is loaded, and the testing images are iterated over to predict labels.\n   - The model predictions for each testing image are printed.\n\n5. **Creating a CSV Submission File:**\n   - The predicted labels for each testing image are stored in a list.\n   - A CSV file (`submission1-1.csv`) is created with the patient ID and corresponding predicted labels.\n   - The CSV file is saved, and the results are printed.\n\n6. **Loading and Displaying the CSV Results:**\n   - The CSV file is loaded into a DataFrame using `pd.read_csv()`.\n   - The contents of the CSV file are printed.\n\nNote: The comments provided in the code have been used for explaining each step briefly.","metadata":{}},{"cell_type":"code","source":"# DEFINING THE LOSS AND OPTIMIZATION\n\n\n#First loss function \n\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=learning_rate)\n\n\n# TRAINING OF MY MODEL\n\nprint_every = 40  # Define how often to print the loss\n\nfor epoch in range(num_epochs):\n    running_loss = 0.0  # Initialize a running loss variable\n    for batch_idx, (images, labels) in enumerate(top_similar_images_loader):\n        #print(f'Batch {batch_idx+1}/{len(top_similar_images_loader)}, Epoch {epoch+1}/{num_epochs}')\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels.float())\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n        #print(f'Loss: {loss.item():.4f}')\n        \n        if (batch_idx + 1) % print_every == 0:  # Print every print_every batches\n            print(f'Epoch [{epoch+1}/{num_epochs}], Batch [{batch_idx+1}/{len(top_similar_images_loader)}], Loss: {running_loss/print_every:.4f}')\n            running_loss = 0.0  # Reset the running loss\n            \n            \n            \n# Saving model\ntorch.save(model.state_dict(), 'model.pth')\n\n# Use Testing images to see the labels\n\nfrom PIL import Image\nimport numpy as np\n\n# Assuming `model` is your trained model\n# Assuming you have defined `CustomModel`\n\n# Load the saved model state\nmodel = CustomModel(num_classes=13)\nmodel.load_state_dict(torch.load('model.pth'))\nmodel.eval()\n\n# Define your image transformation\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.5], std=[0.5])\n])\n\n# Assuming `testing_dir` is your directory containing testing images\ntesting_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/'\nimage_paths=[]\nfor dirname, _, filenames in os.walk(testing_dir):\n    \n    for filename in filenames:\n        \n        image_paths.append(os.path.join(dirname, filename))\n\n# List of file paths for your testing images\n#image_paths = glob(os.path.join(testing_dir, '*.dcm'))\n#print(image_paths)\n# Iterate over each image\nfor image_path in image_paths:\n    print(image_path)\n    dicom_data = pydicom.dcmread(image_path)\n    pixel_data = dicom_data.pixel_array / 255.0\n    image = Image.fromarray(pixel_data)\n    \n    if transform is not None:\n        image = transform(image)\n        image = image.unsqueeze(0)  # Add batch dimension\n    \n    with torch.no_grad():\n        outputs = model(image)\n        labels = (outputs ).float()  # Assuming binary classification\n\n    print(f'Predicted Labels for {image_path}: {labels}')\n    \n\n    \nimport csv\n\n#  Creating the csv file\ntesting_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/'\nimage_paths=[]\nfor dirname, _, filenames in os.walk(testing_dir):\n    \n    for filename in filenames:\n        \n        image_paths.append(os.path.join(dirname, filename))\n\n# List to store the results\nresults = []\n\n# Iterate over each image\nfor image_path in image_paths:\n    print(image_path)\n    dicom_data = pydicom.dcmread(image_path)\n    pixel_data = dicom_data.pixel_array / 255.0\n    image = Image.fromarray(pixel_data)\n    \n    if transform is not None:\n        image = transform(image)\n        image = image.unsqueeze(0)  # Add batch dimension\n    \n    with torch.no_grad():\n        outputs = model(image)\n        labels = (outputs ).float()  # Assuming binary classification\n\n    # Get patient ID from image path\n    patient_id = image_path.split('/')[-2]\n\n    # Convert labels to list for CSV\n    labels_list = labels.squeeze().tolist()\n\n    # Append results\n    results.append([patient_id] + labels_list)\n\n# Define the CSV file path\ncsv_file_path = 'submission1-1.csv'\n\n# Write results to a CSV file\nwith open(csv_file_path, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    # Write the header\n    writer.writerow(['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high'])\n    # Write the data\n    writer.writerows(results)\n\nprint(f'Results saved to {csv_file_path}')\n\n\npd.read_csv('submission1-1.csv')\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2 code with alnother loss function","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DEFINING THE LOSS AND OPTIMIZATION\n\n#criterion = nn.BCEWithLogitsLoss()\n#criterion = nn.HingeEmbeddingLoss()\n#criterion = nn.SmoothL1Loss()\ncriterion = nn.CrossEntropyLoss()\n\n\noptimizer = torch.optim.Adam(model.parameters(), lr=learning_rate)\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TRAINING OF MY MODEL\n\nprint_every = 40  # Define how often to print the loss\n\nfor epoch in range(num_epochs):\n    running_loss = 0.0  # Initialize a running loss variable\n    for batch_idx, (images, labels) in enumerate(top_similar_images_loader):\n        #print(f'Batch {batch_idx+1}/{len(top_similar_images_loader)}, Epoch {epoch+1}/{num_epochs}')\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels.float())\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n        #print(f'Loss: {loss.item():.4f}')\n        \n        if (batch_idx + 1) % print_every == 0:  # Print every print_every batches\n            print(f'Epoch [{epoch+1}/{num_epochs}], Batch [{batch_idx+1}/{len(top_similar_images_loader)}], Loss: {running_loss/print_every:.4f}')\n            running_loss = 0.0  # Reset the running loss","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\n\n#  Creating the csv file\ntesting_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/'\nimage_paths=[]\nfor dirname, _, filenames in os.walk(testing_dir):\n    \n    for filename in filenames:\n        \n        image_paths.append(os.path.join(dirname, filename))\n\n# List to store the results\nresults = []\n\n# Iterate over each image\nfor image_path in image_paths:\n    print(image_path)\n    dicom_data = pydicom.dcmread(image_path)\n    pixel_data = dicom_data.pixel_array / 255.0\n    image = Image.fromarray(pixel_data)\n    \n    if transform is not None:\n        image = transform(image)\n        image = image.unsqueeze(0)  # Add batch dimension\n    \n    with torch.no_grad():\n        outputs = model(image)\n        labels = (outputs ).float()  # Assuming binary classification\n\n    # Get patient ID from image path\n    patient_id = image_path.split('/')[-2]\n\n    # Convert labels to list for CSV\n    labels_list = labels.squeeze().tolist()\n\n    # Append results\n    results.append([patient_id] + labels_list)\n\n# Define the CSV file path\ncsv_file_path = 'submission1-2.csv'\n\n# Write results to a CSV file\nwith open(csv_file_path, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    # Write the header\n    writer.writerow(['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high'])\n    # Write the data\n    writer.writerows(results)\n\nprint(f'Results saved to {csv_file_path}')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission1-2.csv')\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  3 code with alnother loss function\n\n# DEFINING THE LOSS AND OPTIMIZATION\n\n#criterion = nn.BCEWithLogitsLoss()\n#criterion = nn.CrossEntropyLoss()\ncriterion = nn.HingeEmbeddingLoss()\n#criterion = nn.SmoothL1Loss()\n\n\n\noptimizer = torch.optim.Adam(model.parameters(), lr=learning_rate)\n\n\n\n# TRAINING OF MY MODEL\n\nprint_every = 40  # Define how often to print the loss\n\nfor epoch in range(num_epochs):\n    running_loss = 0.0  # Initialize a running loss variable\n    for batch_idx, (images, labels) in enumerate(top_similar_images_loader):\n        #print(f'Batch {batch_idx+1}/{len(top_similar_images_loader)}, Epoch {epoch+1}/{num_epochs}')\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels.float())\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n        #print(f'Loss: {loss.item():.4f}')\n        \n        if (batch_idx + 1) % print_every == 0:  # Print every print_every batches\n            print(f'Epoch [{epoch+1}/{num_epochs}], Batch [{batch_idx+1}/{len(top_similar_images_loader)}], Loss: {running_loss/print_every:.4f}')\n            running_loss = 0.0  # Reset the running loss\n\n            \n            \nimport csv\n\n#  Creating the csv file\ntesting_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/'\nimage_paths=[]\nfor dirname, _, filenames in os.walk(testing_dir):\n    \n    for filename in filenames:\n        \n        image_paths.append(os.path.join(dirname, filename))\n\n# List to store the results\nresults = []\n\n# Iterate over each image\nfor image_path in image_paths:\n    print(image_path)\n    dicom_data = pydicom.dcmread(image_path)\n    pixel_data = dicom_data.pixel_array / 255.0\n    image = Image.fromarray(pixel_data)\n    \n    if transform is not None:\n        image = transform(image)\n        image = image.unsqueeze(0)  # Add batch dimension\n    \n    with torch.no_grad():\n        outputs = model(image)\n        labels = (outputs ).float()  # Assuming binary classification\n\n    # Get patient ID from image path\n    patient_id = image_path.split('/')[-2]\n\n    # Convert labels to list for CSV\n    labels_list = labels.squeeze().tolist()\n\n    # Append results\n    results.append([patient_id] + labels_list)\n\n# Define the CSV file path\ncsv_file_path = 'submission1-3.csv'\n\n# Write results to a CSV file\nwith open(csv_file_path, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    # Write the header\n    writer.writerow(['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high'])\n    # Write the data\n    writer.writerows(results)\n\nprint(f'Results saved to {csv_file_path}')\n\npd.read_csv('submission1-3.csv')\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  4 code with alnother loss function\n\n# DEFINING THE LOSS AND OPTIMIZATION\n\n#criterion = nn.BCEWithLogitsLoss()\n#criterion = nn.CrossEntropyLoss()\n#criterion = nn.HingeEmbeddingLoss()\n# nn.SmoothL1Loss()\n\n\ncriterion = nn.BCELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=learning_rate)\n\n\n\n# TRAINING OF MY MODEL\n\nprint_every = 40  # Define how often to print the loss\n\nfor epoch in range(num_epochs):\n    running_loss = 0.0  # Initialize a running loss variable\n    for batch_idx, (images, labels) in enumerate(top_similar_images_loader):\n        #print(f'Batch {batch_idx+1}/{len(top_similar_images_loader)}, Epoch {epoch+1}/{num_epochs}')\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels.float())\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n        #print(f'Loss: {loss.item():.4f}')\n        \n        if (batch_idx + 1) % print_every == 0:  # Print every print_every batches\n            print(f'Epoch [{epoch+1}/{num_epochs}], Batch [{batch_idx+1}/{len(top_similar_images_loader)}], Loss: {running_loss/print_every:.4f}')\n            running_loss = 0.0  # Reset the running loss\n\n            \n            \nimport csv\n\n#  Creating the csv file\ntesting_dir = '/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/'\nimage_paths=[]\nfor dirname, _, filenames in os.walk(testing_dir):\n    \n    for filename in filenames:\n        \n        image_paths.append(os.path.join(dirname, filename))\n\n# List to store the results\nresults = []\n\n# Iterate over each image\nfor image_path in image_paths:\n    print(image_path)\n    dicom_data = pydicom.dcmread(image_path)\n    pixel_data = dicom_data.pixel_array / 255.0\n    image = Image.fromarray(pixel_data)\n    \n    if transform is not None:\n        image = transform(image)\n        image = image.unsqueeze(0)  # Add batch dimension\n    \n    with torch.no_grad():\n        outputs = model(image)\n        labels = (outputs ).float()  # Assuming binary classification\n\n    # Get patient ID from image path\n    patient_id = image_path.split('/')[-2]\n\n    # Convert labels to list for CSV\n    labels_list = labels.squeeze().tolist()\n\n    # Append results\n    results.append([patient_id] + labels_list)\n\n# Define the CSV file path\ncsv_file_path = 'submission1-4.csv'\n\n# Write results to a CSV file\nwith open(csv_file_path, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    # Write the header\n    writer.writerow(['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high'])\n    # Write the data\n    writer.writerows(results)\n\nprint(f'Results saved to {csv_file_path}')\n\npd.read_csv('submission1-4.csv')\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}