{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070},{"sourceType":"datasetVersion","sourceId":7143819,"datasetId":4123500,"databundleVersionId":7232345},{"sourceType":"datasetVersion","sourceId":6787036,"datasetId":3905056,"databundleVersionId":6872022},{"sourceType":"datasetVersion","sourceId":34674,"datasetId":26069,"databundleVersionId":35697},{"sourceType":"datasetVersion","sourceId":7145869,"datasetId":4125000,"databundleVersionId":7234598},{"sourceType":"datasetVersion","sourceId":7145758,"datasetId":4124921,"databundleVersionId":7234482},{"sourceType":"datasetVersion","sourceId":7145833,"datasetId":4124971,"databundleVersionId":7234557}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport pydicom\nimport os\nimport SimpleITK as sitk\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport cv2\n!pip install kaggle\n\n# Load the CSV file with annotations\nannotations_csv = r\"/kaggle/input/brain-csv/1_Initial_Manual_Labeling.csv\"  # Replace with the path to your CSV file\nannotations_df = pd.read_csv(annotations_csv)\n\n\n# Directory containing DICOM images from the CQ500 dataset\ndicom_dir = r\"/kaggle/input/qureai-headct\"\n\n# Create a dictionary to map SOP Instance UIDs to DICOM file paths\nsop_uid_to_filepath = {}\nfor root, dirs, files in os.walk(dicom_dir):\n    for file in files:\n        if file.endswith(\".dcm\"):\n            dicom_file_path = os.path.join(root, file)\n            dicom_image = pydicom.dcmread(dicom_file_path)\n            sop_uid = dicom_image.SOPInstanceUID\n            sop_uid_to_filepath[sop_uid] = dicom_file_path\n\ni = 0            \nimage_label_mapping = {}\nimage_array = []\nlabel_array = []\nfor index, row in annotations_df.iterrows():\n    sop_uid = row['SOPInstanceUID']\n    #print(sop_uid)\n    if sop_uid in sop_uid_to_filepath:\n        dicom_image_path = sop_uid_to_filepath[sop_uid]\n        dicom_image = pydicom.dcmread(dicom_image_path)\n        ct_image = sitk.ReadImage(sop_uid_to_filepath[sop_uid])\n        image_label_mapping[sop_uid] = (dicom_image, row)\n        label_array.append(row)\n        ct_array = sitk.GetArrayFromImage(ct_image)\n        image_array.append(ct_array)\n        i += 1\nprint(i, 'Done')","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:42:51.644943Z","iopub.execute_input":"2023-12-08T04:42:51.645443Z","iopub.status.idle":"2023-12-08T04:43:12.133771Z","shell.execute_reply.started":"2023-12-08T04:42:51.645405Z","shell.execute_reply":"2023-12-08T04:43:12.130056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-processing  the image array\nimport numpy as np\nimport cv2\n\ndef pre_process_im_arr(image_array):\n    # Define window levels and widths for brain tissue\n    window_level = 40  # Adjust this value based on your data\n    window_width = 80  # Adjust this value based on your data\n\n    # Apply windowing to the CT scan\n    min_value = window_level - window_width / 2\n    max_value = window_level + window_width / 2\n    image_array[image_array < min_value] = min_value\n    image_array[image_array > max_value] = max_value\n\n    # Normalize the HU values to the range [0, 255]\n    image_array = ((image_array - min_value) / (max_value - min_value) * 255).astype(np.uint8)\n\n    # Apply morphology dilation to remove noise\n    kernel = np.ones((3, 3), np.uint8)  # Adjust the kernel size as needed\n    dilated_image = cv2.dilate(image_array, kernel, iterations=1)\n\n    # Squeeze the extra dimension to convert it into a 2D array\n    dilated_image = dilated_image.squeeze()\n\n    return dilated_image\n\nfor i in range (len(image_array)):\n    image_array[i] = pre_process_im_arr(image_array[i])","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.134966Z","iopub.status.idle":"2023-12-08T04:43:12.135422Z","shell.execute_reply.started":"2023-12-08T04:43:12.135214Z","shell.execute_reply":"2023-12-08T04:43:12.135235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the directory to save the PNG images\noutput_dir = r\"/kaggle/working/Image\"\n\n# Ensure the output directory exists, or create it\nif not os.path.exists(output_dir):\n    os.makedirs(output_dir)\n\n# Loop through the pre-processed images in image_array and save them as PNG\nfor i, processed_image in enumerate(image_array):\n    sop_uid = label_array[i]['SOPInstanceUID']\n\n    # Resizing the image to 640x640\n    processed_image = cv2.resize(processed_image, (640, 640))\n    \n    # Rename the output file as \"image{i}.jpg\"\n    output_path = os.path.join(output_dir, f\"image{i}.jpg\")\n    \n    # Save the pre-processed image as a jpg\n    cv2.imwrite(output_path, processed_image)\n\nprint(f\"{i+1} pre-processed images saved as jpg in {output_dir}\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.136905Z","iopub.status.idle":"2023-12-08T04:43:12.13832Z","shell.execute_reply.started":"2023-12-08T04:43:12.138072Z","shell.execute_reply":"2023-12-08T04:43:12.1381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lists to store annotations and hemorrhage type labels\nannotations = []\nhemorrhage_labels = []\n\n# Iterate through the 'labels' array, which contains both types of labels\nfor label_row in label_array:\n    # Append the annotation data to the 'annotations' list\n    annotations.append(label_row['data'])  # Replace 'Annotation_Column' with the actual column name for annotations\n    # Append the hemorrhage type label to the 'hemorrhage_labels' list\n    hemorrhage_labels.append(label_row['labelName'])  # Replace 'Hemorrhage_Type_Column' with the actual column name for hemorrhage types","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.139339Z","iopub.status.idle":"2023-12-08T04:43:12.140215Z","shell.execute_reply.started":"2023-12-08T04:43:12.140003Z","shell.execute_reply":"2023-12-08T04:43:12.140025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(annotations[0])\nprint(hemorrhage_labels[0])","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.141338Z","iopub.status.idle":"2023-12-08T04:43:12.141753Z","shell.execute_reply.started":"2023-12-08T04:43:12.141558Z","shell.execute_reply":"2023-12-08T04:43:12.141578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels = []\nfor i in range(len(hemorrhage_labels)):\n    if hemorrhage_labels[i] == 'Intraparenchymal':\n        num_labels.append(0)\n    elif hemorrhage_labels[i] == 'Intraventricular':\n        num_labels.append(1)\n    elif hemorrhage_labels[i] == 'Subarachnoid':\n        num_labels.append(2)\n    elif hemorrhage_labels[i] == 'Epidural':\n        num_labels.append(3)\n    elif hemorrhage_labels[i] == 'Subdural':\n        num_labels.append(4)\n    elif hemorrhage_labels[i] == 'Chronic':\n        num_labels.append(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.142922Z","iopub.status.idle":"2023-12-08T04:43:12.143332Z","shell.execute_reply.started":"2023-12-08T04:43:12.143135Z","shell.execute_reply":"2023-12-08T04:43:12.143154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(num_labels[0])\nprint(annotations[0])","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.144709Z","iopub.status.idle":"2023-12-08T04:43:12.14517Z","shell.execute_reply.started":"2023-12-08T04:43:12.144965Z","shell.execute_reply":"2023-12-08T04:43:12.144986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport pandas as pd\n\n# Your list of JSON strings\njson_strings = annotations\n\n# Initialize lists to store the parsed values\nx_values = []\ny_values = []\nwidth_values = []\nheight_values = []\n\n# Parse the JSON strings and extract values\nfor json_str in json_strings:\n    data = json.loads(json_str.replace(\"'\", \"\\\"\"))  # Ensure single quotes are replaced with double quotes\n    x_values.append(data['x'])\n    y_values.append(data['y'])\n    width_values.append(data['width'])\n    height_values.append(data['height'])\n\n# Create a DataFrame to store the values\ndf = pd.DataFrame({\n    'x': x_values,\n    'y': y_values,\n    'width': width_values,\n    'height': height_values\n})\n\n# Save the DataFrame to a CSV file\ndf.to_csv('annotations.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.146473Z","iopub.status.idle":"2023-12-08T04:43:12.146942Z","shell.execute_reply.started":"2023-12-08T04:43:12.146685Z","shell.execute_reply":"2023-12-08T04:43:12.146704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotations_df = pd.read_csv('annotations.csv')\n\noutput_dir = r\"/kaggle/working/Lable\"\n\nif not os.path.exists(output_dir):\n    os.makedirs(output_dir)\n\nfor i, (_, row) in enumerate(annotations_df.iterrows()):\n    x = row['x']\n    y = row['y']\n    width = row['width']\n    height = row['height']\n\n    label_new = [num_labels[i], (x+width/2)/512, (y+height/2)/512, width/512, height/512]\n    label_string = ' '.join(map(str, label_new))  # Convert elements to strings and join with spaces\n\n    # Define the unique file name for each label_string (e.g., using index i)\n    output_file = os.path.join(output_dir, f\"image{i}.txt\")\n\n    with open(output_file, 'w') as file:\n        file.write(label_string)\n\nprint(f\"Values saved to {output_dir}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.148238Z","iopub.status.idle":"2023-12-08T04:43:12.148644Z","shell.execute_reply.started":"2023-12-08T04:43:12.148458Z","shell.execute_reply":"2023-12-08T04:43:12.148477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nfrom sklearn.model_selection import train_test_split\n\n# Path to the output directory in Kaggle\noutput_directory = r\"/kaggle/working/\"\nimage_dir = r\"/kaggle/working/Image\"\n\n# Define the directory where your label text files are located\nlabel_dir = r\"/kaggle/working/Lable\"\n\n# Define the directories for saving training and validation images\ntrain_image_dir = r\"/kaggle/working/trainimg/\"\nval_image_dir = r\"/kaggle/working/valimg/\"\n\n# Define the directories for saving training and validation labels\ntrain_label_dir = r\"/kaggle/working/trainlbl/\"\nval_label_dir = r\"/kaggle/working/vallbl/\"\nif not os.path.exists(train_image_dir):\n    os.makedirs(train_image_dir)\nif not os.path.exists(val_image_dir):\n    os.makedirs(val_image_dir)\nif not os.path.exists(train_label_dir):\n    os.makedirs(train_label_dir)\nif not os.path.exists(val_label_dir):\n    os.makedirs(val_label_dir)\n# List all the image files and label files\nimage_files = [f for f in os.listdir(image_dir) if f.endswith(\".jpg\")]\nlabel_files = [f for f in os.listdir(label_dir) if f.endswith(\".txt\")]\nprint(len(image_files))\nprint(len(label_files))\n\n# Ensure that the lists are sorted for consistency\nimage_files.sort()\nlabel_files.sort()\n\n# Split the dataset into training and validation sets\ntrain_images, val_images, train_labels, val_labels = train_test_split(\n    image_files, label_files, test_size=0.5, random_state=42\n)\nprint(train_labels)\n# Print the number of training and validation images\nprint(f\"Number of training images: {len(train_images)}\")\nprint(f\"Number of validation images: {len(val_images)}\")\n\n# Copy training images to the training directory\nfor image in train_images:\n    source_path = os.path.join(image_dir, image)\n    destination_path = os.path.join(train_image_dir, image)\n    shutil.copy(source_path, destination_path)\n\n# Copy validation images to the validation directory\nfor image in val_images:\n    source_path = os.path.join(image_dir, image)\n    destination_path = os.path.join(val_image_dir, image)\n    shutil.copy(source_path, destination_path)\n\n# Copy training labels to the training directory\nfor label in train_labels:\n    source_path = os.path.join(label_dir, label)\n    destination_path = os.path.join(train_label_dir, label)\n    shutil.copy(source_path, destination_path)\n\n# Copy validation labels to the validation directory\nfor label in val_labels:\n    source_path = os.path.join(label_dir, label)\n    destination_path = os.path.join(val_label_dir, label)\n    shutil.copy(source_path, destination_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.150498Z","iopub.status.idle":"2023-12-08T04:43:12.151094Z","shell.execute_reply.started":"2023-12-08T04:43:12.15077Z","shell.execute_reply":"2023-12-08T04:43:12.150817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot images with bounding boxes\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\n\n# Define the directory containing the images\n#image_dir = r\"E:\\UOM\\Academic\\Semester 5\\Image processing and machine vision\\Project\\Outputs\\output_png_images\"\n\n# Define the directory containing the labels\n#label_dir = r\"E:\\UOM\\Academic\\Semester 5\\Image processing and machine vision\\Project\\Outputs\\output_labels\"\n\n# Plot the first 10 images\nfor i in range(6):\n    # Read the image\n    image_path = os.path.join(image_dir, f\"image{i}.jpg\")\n    image = cv2.imread(image_path)\n\n    # Read the label\n    label_path = os.path.join(label_dir, f\"image{i}.txt\")\n    with open(label_path, 'r') as file:\n        label = file.read()\n\n    # Parse the label string\n    label = label.split()\n    label = [float(x) for x in label]\n\n    # Extract the label values\n    class_id = int(label[0])\n    x = label[1] * 640\n    y = label[2] * 640\n    width = label[3] * 640\n    height = label[4] * 640\n\n    # Calculate the top-left corner coordinates\n    left = x - width / 2\n    top = y - height / 2\n\n    # Draw the bounding box\n    cv2.rectangle(image, (int(left), int(top)), (int(left + width), int(top + height)), (255, 0, 0), 2)\n\n    # Convert the image from BGR (OpenCV format) to RGB\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    # Plot the image\n    print(f\"Image {i}\")\n    plt.imshow(image)\n    plt.show()\n    print(image.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.152738Z","iopub.status.idle":"2023-12-08T04:43:12.153198Z","shell.execute_reply.started":"2023-12-08T04:43:12.152985Z","shell.execute_reply":"2023-12-08T04:43:12.153005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n\n\nimage_dir = \"/kaggle/working/Image\"\nlabel_dir = \"/kaggle/working/Lable\"\n\n# Function to load image paths and corresponding labels\ndef load_data(image_dir, label_dir):\n    image_paths = []\n    labels = []\n\n    for filename in os.listdir(image_dir):\n        if filename.endswith(\".jpg\"):  # Adjust the extension based on your image format\n            image_path = os.path.join(image_dir, filename)\n            label_path = os.path.join(label_dir, filename.replace(\".jpg\", \".txt\"))\n\n            if os.path.exists(label_path):\n                with open(label_path, 'r') as label_file:\n                    label = int(label_file.read(1))  # Read the first character as an integer\n                    image_paths.append(image_path)\n                    labels.append(str(label))  # Convert to string\n\n    return np.array(image_paths), np.array(labels)\n\n# Load image paths and labels\nimage_paths, labels = load_data(image_dir, label_dir)\n\n# Split the data into training and validation sets\ntrain_image_paths, val_image_paths, train_labels, val_labels = train_test_split(\n    image_paths, labels, test_size=0.2, random_state=42\n)\n\n# Create an ImageDataGenerator for data augmentation and normalization\nbatch_size = 32\nimg_height, img_width = 224, 224\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True\n)\n\n# Create flow_from_directory generators for training and validation data\ntrain_generator = train_datagen.flow_from_dataframe(\n    pd.DataFrame({\"image_paths\": train_image_paths, \"labels\": train_labels}),\n    x_col=\"image_paths\",\n    y_col=\"labels\",\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='sparse',  # 'sparse' for integer labels\n)\n\nval_generator = train_datagen.flow_from_dataframe(\n    pd.DataFrame({\"image_paths\": val_image_paths, \"labels\": val_labels}),\n    x_col=\"image_paths\",\n    y_col=\"labels\",\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='sparse',  # 'sparse' for integer labels\n)\n\n# Create the ResNet model\nbase_model = tf.keras.applications.ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\nmodel = models.Sequential([\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dense(256, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(7, activation='softmax')  # Assuming 7 classes (0 to 6)\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(train_generator, epochs=100, validation_data=val_generator)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.154754Z","iopub.status.idle":"2023-12-08T04:43:12.155203Z","shell.execute_reply.started":"2023-12-08T04:43:12.155Z","shell.execute_reply":"2023-12-08T04:43:12.155019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing import image\n\n# Function to predict and plot images\ndef predict_and_plot(model, generator, num_images=5):\n    # Get a batch of images and their true labels\n    images, true_labels = next(generator)\n\n    # Make predictions\n    predictions = model.predict(images)\n\n    # Get class labels\n    class_labels = list(generator.class_indices.keys())\n\n    # Plot the images with true and predicted labels\n    plt.figure(figsize=(15, 8))\n    for i in range(num_images):\n        plt.subplot(1, num_images, i+1)\n        plt.imshow(images[i])\n        plt.title(f\"True: {class_labels[int(true_labels[i])]}, Predicted: {class_labels[np.argmax(predictions[i])]}\")\n        plt.axis('off')\n\n    plt.show()\n\n# Use the trained model to make predictions on a batch of validation images\npredict_and_plot(model, val_generator)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.156154Z","iopub.status.idle":"2023-12-08T04:43:12.156533Z","shell.execute_reply.started":"2023-12-08T04:43:12.156347Z","shell.execute_reply":"2023-12-08T04:43:12.156365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing import image\n\n# Function to predict and plot images with label names\ndef predict_and_plot_with_names(model, generator, label_mapping, num_images=5):\n    # Get a batch of images and their true labels\n    images, true_labels = next(generator)\n\n    # Make predictions\n    predictions = model.predict(images)\n\n    # Get class labels\n    class_labels = list(generator.class_indices.keys())\n\n    # Plot the images with true and predicted labels\n    plt.figure(figsize=(8, 8))\n    for i in range(num_images):\n        plt.subplot(1, num_images, i+1)\n        plt.imshow(images[i])\n        \n        # Convert numerical labels to names\n        true_label_name = label_mapping[int(true_labels[i])]\n        predicted_label_name = label_mapping[np.argmax(predictions[i])]\n        \n        plt.title(f\"True: {true_label_name}, Predicted: {predicted_label_name}\")\n        plt.axis('off')\n\n    plt.show()\n\n# Define your mapping\nlabel_mapping = {\n    0: 'Intraparenchymal',\n    1: 'Intraventricular',\n    2: 'Subarachnoid',\n    3: 'Epidural',\n    4: 'Subdural',\n    5: 'Chronic'\n}\n\n# Use the trained model to make predictions on a batch of validation images with label names\npredict_and_plot_with_names(model, val_generator, label_mapping)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.157678Z","iopub.status.idle":"2023-12-08T04:43:12.158071Z","shell.execute_reply.started":"2023-12-08T04:43:12.157881Z","shell.execute_reply":"2023-12-08T04:43:12.157898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing import image\n\n# Function to predict and plot images with label names in a column\ndef predict_and_plot_in_column(model, generator, label_mapping, num_images=5):\n    # Get a batch of images and their true labels\n    images, true_labels = next(generator)\n\n    # Make predictions\n    predictions = model.predict(images)\n\n    # Get class labels\n    class_labels = list(generator.class_indices.keys())\n\n    # Plot the images with true and predicted labels in a column\n    plt.figure(figsize=(8, 15))  # Adjust figsize based on your preference\n    for i in range(num_images):\n        plt.subplot(num_images, 1, i+1)  # Arrange images in a single column\n        plt.imshow(images[i])\n        \n        # Convert numerical labels to names\n        true_label_name = label_mapping[int(true_labels[i])]\n        predicted_label_name = label_mapping[np.argmax(predictions[i])]\n        \n        plt.title(f\"True: {true_label_name}, Predicted: {predicted_label_name}\")\n        plt.axis('off')\n\n    plt.show()\n\n# Use the trained model to make predictions on a batch of validation images with label names in a column\npredict_and_plot_in_column(model, val_generator, label_mapping)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T04:43:12.159027Z","iopub.status.idle":"2023-12-08T04:43:12.159376Z","shell.execute_reply.started":"2023-12-08T04:43:12.159201Z","shell.execute_reply":"2023-12-08T04:43:12.159218Z"},"trusted":true},"execution_count":null,"outputs":[]}]}