{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-01T02:38:25.681864Z","iopub.execute_input":"2024-06-01T02:38:25.682288Z","iopub.status.idle":"2024-06-01T02:38:25.688554Z","shell.execute_reply.started":"2024-06-01T02:38:25.682253Z","shell.execute_reply":"2024-06-01T02:38:25.687096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/input","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:25.690984Z","iopub.execute_input":"2024-06-01T02:38:25.692066Z","iopub.status.idle":"2024-06-01T02:38:26.757758Z","shell.execute_reply.started":"2024-06-01T02:38:25.692031Z","shell.execute_reply":"2024-06-01T02:38:26.756228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONVERT_DICOM_TO_PNG = True\nUPLOAD_TO_KAGGLE = True","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:26.759565Z","iopub.execute_input":"2024-06-01T02:38:26.759926Z","iopub.status.idle":"2024-06-01T02:38:26.765227Z","shell.execute_reply.started":"2024-06-01T02:38:26.759893Z","shell.execute_reply":"2024-06-01T02:38:26.764113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CONVERT_DICOM_TO_PNG:\n    !pip install SimpleITK # converting the images from dicom to png for use with yolo","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:26.768047Z","iopub.execute_input":"2024-06-01T02:38:26.768459Z","iopub.status.idle":"2024-06-01T02:38:39.20482Z","shell.execute_reply.started":"2024-06-01T02:38:26.768428Z","shell.execute_reply":"2024-06-01T02:38:39.203353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!(ls /kaggle/temp || mkdir /kaggle/temp) && rm -rf /kaggle/temp/* # create temp dir if it doesn't exist already and clear its contents","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:39.206823Z","iopub.execute_input":"2024-06-01T02:38:39.207248Z","iopub.status.idle":"2024-06-01T02:38:40.27728Z","shell.execute_reply.started":"2024-06-01T02:38:39.20721Z","shell.execute_reply":"2024-06-01T02:38:40.275853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ","metadata":{}},{"cell_type":"code","source":"import logging\nDEBUG = True\nTESTING = False\n\n# Remove all handlers associated with the root logger object.\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\nif DEBUG:\n    logging.basicConfig(level=logging.DEBUG) # explicitly set the logging level to debug\nelse:\n    logging.basicConfig(level=logging.ERROR) # explicitly set the logging level to error","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:40.279091Z","iopub.execute_input":"2024-06-01T02:38:40.279493Z","iopub.status.idle":"2024-06-01T02:38:40.287506Z","shell.execute_reply.started":"2024-06-01T02:38:40.279453Z","shell.execute_reply":"2024-06-01T02:38:40.286067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport os,sys\nfrom pydicom import read_file\nimport tqdm\nfrom PIL.Image import fromarray\nfrom pathlib import Path\n\nimport csv\nimport functools\nimport itertools\n\nfrom multiprocessing import cpu_count\nfrom tqdm.contrib.concurrent import process_map\nimport multiprocessing\n\nimport SimpleITK as sitk\n\n# for multiprocessing\nN_PROCESSES = cpu_count()\nprint(f\"Using {N_PROCESSES} processes out of {cpu_count()} cores.\")\n\nN_IMAGES_TO_USE = 15000\n\nSOURCE_FOLDER = r'/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train'\nOUTPUT_FOLDER = r'/kaggle/temp/input_images'\n\ndef convert_image(input_file_name, output_file_name, source_folder=SOURCE_FOLDER, output_folder=OUTPUT_FOLDER, new_width=None):\n    try:\n        image_file_reader = sitk.ImageFileReader()\n        # only read DICOM images\n        image_file_reader.SetImageIO(\"GDCMImageIO\")\n        image_file_reader.SetFileName(str(Path(source_folder).joinpath(input_file_name)))\n        image_file_reader.ReadImageInformation()\n        image_size = list(image_file_reader.GetSize())\n        if len(image_size) == 3 and image_size[2] == 1:\n            image_size[2] = 0\n        image_file_reader.SetExtractSize(image_size)\n        image = image_file_reader.Execute()\n        if new_width:\n            original_size = image.GetSize()\n            original_spacing = image.GetSpacing()\n            new_spacing = [\n                (original_size[0] - 1) * original_spacing[0] / (new_width - 1)\n            ] * 2\n            new_size = [\n                new_width,\n                int((original_size[1] - 1) * original_spacing[1] / new_spacing[1]),\n            ]\n            image = sitk.Resample(\n                image1=image,\n                size=new_size,\n                transform=sitk.Transform(),\n                interpolator=sitk.sitkLinear,\n                outputOrigin=image.GetOrigin(),\n                outputSpacing=new_spacing,\n                outputDirection=image.GetDirection(),\n                defaultPixelValue=0,\n                outputPixelType=image.GetPixelID(),\n            )\n        # If a single channel image, rescale to [0,255]. Also modify the\n        # intensity values based on the photometric interpretation. If\n        # MONOCHROME2 (minimum should be displayed as black) we don't need to\n        # do anything, if image has MONOCRHOME1 (minimum should be displayed as\n        # white) we flip # the intensities. This is a constraint imposed by ITK\n        # which always assumes MONOCHROME2.\n        if image.GetNumberOfComponentsPerPixel() == 1:\n            image = sitk.RescaleIntensity(image, 0, 255)\n            if image_file_reader.GetMetaData(\"0028|0004\").strip() == \"MONOCHROME1\":\n                image = sitk.InvertIntensity(image, maximum=255)\n            image = sitk.Cast(image, sitk.sitkUInt8)\n        sitk.WriteImage(image, str(Path(output_folder).joinpath(output_file_name)))\n            \n        del image  # Explicitly delete the image to free memory\n        gc.collect()  # Force garbage collection\n                \n        return True\n    except BaseException:\n        return False\n\n    \ndef convert_images(source_folder, output_folder, n_images = None, new_width = None):\n    if n_images is None:\n        input_file_names = os.listdir(source_folder)\n    else:\n        input_file_names = os.listdir(source_folder)[:n_images]\n    logging.debug(f\"Sample of input file names: {input_file_names[:5]}\")\n    output_file_extension = \".png\"\n    output_file_names = [os.path.splitext(os.path.basename(file_name))[0] + output_file_extension for file_name in input_file_names]\n    logging.debug(f\"Sample of output file names: {output_file_names[:5]}\")\n\n    total_entries = len(input_file_names)\n    \n    with multiprocessing.Pool(processes=N_PROCESSES) as pool:\n        return input_file_names, output_file_names, pool.starmap(\n            functools.partial(convert_image, source_folder=source_folder, output_folder=output_folder, new_width=new_width),\n            zip(input_file_names, output_file_names),\n        )\n\nif CONVERT_DICOM_TO_PNG: \n    if not os.path.exists(OUTPUT_FOLDER):\n        os.mkdir(OUTPUT_FOLDER)\n    input_file_names, output_file_names, res = convert_images(SOURCE_FOLDER, OUTPUT_FOLDER, N_IMAGES_TO_USE)\n    \n    input_file_names = list(itertools.compress(input_file_names, res))\n    output_file_names = list(itertools.compress(output_file_names, res))","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:40.289708Z","iopub.execute_input":"2024-06-01T02:38:40.290172Z","iopub.status.idle":"2024-06-01T02:38:59.431139Z","shell.execute_reply.started":"2024-06-01T02:38:40.29014Z","shell.execute_reply":"2024-06-01T02:38:59.42951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Checking some images","metadata":{}},{"cell_type":"code","source":"import cv2\nimport glob\nimport matplotlib.pyplot as plt\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\nimages_path = glob.glob(str(OUTPUT_FOLDER) + \"/*\")[25:50] ## SaveImagePath\n\nimg_array = [cv2.imread(image_path) for image_path in images_path]\nplot_imgs(img_array)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count the number of images\n!find /kaggle/temp/input_images -type f -name '*.png' -printf x | wc -c\n","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:38:59.452901Z","iopub.execute_input":"2024-06-01T02:38:59.453258Z","iopub.status.idle":"2024-06-01T02:39:00.475202Z","shell.execute_reply.started":"2024-06-01T02:38:59.453229Z","shell.execute_reply":"2024-06-01T02:39:00.473873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create a YOLO-v8 compatible dataset","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nimport csv\nimport os, sys\nimport shutil\nimport tqdm\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport cv2\n\nCOPY_IMAGES = True\nDELETE_IMAGES_FROM_PREVIOUS_SOURCE = CONVERT_DICOM_TO_PNG # if there is not conversion done, then there is no need to move\n\n\nINPUT_DIR = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection\")\nINPUT_CSV = INPUT_DIR.joinpath(\"train.csv\")\nif CONVERT_DICOM_TO_PNG:\n    INPUT_IMAGES = Path(OUTPUT_FOLDER)\nelse:\n    INPUT_IMAGES = INPUT_DIR.joinpath(\"train\")\n    \n# restricting to these image ids\npresent_image_ids = os.listdir(INPUT_IMAGES)\n\nTRAIN_TEST_VALIDATION_SPLIT = (70, 20, 10)\n\nOUTPUT_DIR = Path(\"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO\")\n\nTRAIN_OUTPUT_IMAGES = OUTPUT_DIR.joinpath(\"train\", \"images\")\nTEST_OUTPUT_IMAGES = OUTPUT_DIR.joinpath(\"test\", \"images\")\nVAL_OUTPUT_IMAGES = OUTPUT_DIR.joinpath(\"val\", \"images\")\n\nTRAIN_OUTPUT_TXTS = OUTPUT_DIR.joinpath(\"train\", \"labels\")\nTEST_OUTPUT_TXTS = OUTPUT_DIR.joinpath(\"test\", \"labels\")\nVAL_OUTPUT_TXTS = OUTPUT_DIR.joinpath(\"val\", \"labels\")\n\n# for cross checking\nCLASS_NAME_AND_ID = {\n    0: \"Aortic enlargement\",\n    1: \"Atelectasis\",\n    2: \"Calcification\",\n    3: \"Cardiomegaly\",\n    4: \"Consolidation\",\n    5: \"ILD\",\n    6: \"Infiltration\",\n    7: \"Lung Opacity\",\n    8: \"Nodule/Mass\",\n    9: \"Other lesion\",\n    10: \"Pleural effusion\",\n    11: \"Pleural thickening\",\n    12: \"Pneumothorax\",\n    13: \"Pulmonary fibrosis\",\n    14: \"No finding\",\n}\n\nlogging.debug(f\"Input Directory: {INPUT_DIR.__str__()}\")\nlogging.debug(f\"Output Directory: {OUTPUT_DIR.__str__()}\")\n\n\n# create output folder if it doesn't exist already\nif not os.path.exists(OUTPUT_DIR):\n    logging.debug(f\"Creating the directory {OUTPUT_DIR.__str__()} as it didn't exist.\")\n    OUTPUT_DIR.mkdir(parents=False, exist_ok=True)\n\nif COPY_IMAGES:\n    for output_images_folder in [TRAIN_OUTPUT_IMAGES, TEST_OUTPUT_IMAGES, VAL_OUTPUT_IMAGES]:\n        output_images_folder.mkdir(parents=True, exist_ok=True) # create the output images folder (just moving)\n\nfor output_text_folder in [TRAIN_OUTPUT_TXTS, TEST_OUTPUT_TXTS, VAL_OUTPUT_TXTS]:\n    output_text_folder.mkdir(parents=True, exist_ok=True) # create the labels folder too\n\n# Read the CSV and get unique image IDs\ndf = pd.read_csv(INPUT_CSV.__str__())\nunique_image_ids = df[\"image_id\"].unique()\n\n# finding the image ids for train, test and validation split\n# First split to get train and remaining (test + val) sets\ntrain_image_ids, remaining_image_ids = train_test_split(\n    unique_image_ids, test_size=sum(TRAIN_TEST_VALIDATION_SPLIT[1:])/sum(TRAIN_TEST_VALIDATION_SPLIT), random_state=42\n)\n\n# Second split to get test and val sets from the remaining data\nval_image_ids, test_image_ids = train_test_split(\n    remaining_image_ids, test_size=float(TRAIN_TEST_VALIDATION_SPLIT[2])/sum(TRAIN_TEST_VALIDATION_SPLIT[1:]), random_state=42\n)\n\n# Print the number of samples in each set\nlogging.info(f'Train set size: {len(train_image_ids)}')\nlogging.info(f'Validation set size: {len(val_image_ids)}')\nlogging.info(f'Test set size: {len(test_image_ids)}')\n\n# each line is in the format image_id,class_name,class_id,rad_id,x_min,y_min,x_max,y_max\nif TESTING:\n    n_max = 5\n\nimg_extension = \".png\" if CONVERT_DICOM_TO_PNG else \".jpg\"\n    \nwith open(INPUT_CSV.__str__(), mode=\"r\") as input_csv_file:\n    for i, line in tqdm.tqdm(enumerate(csv.reader(input_csv_file, delimiter=\"\\n\"))):\n        if i == 0:  # skip the header\n            continue\n\n        values = line[0].split(\",\")\n\n        image_id, class_name, class_id, _, x_min, y_min, x_max, y_max = values\n        \n        image_id_with_extension = str(image_id + img_extension)\n        if image_id_with_extension not in present_image_ids: # skip the absent image ids\n            continue\n        \n        expected_class_name = CLASS_NAME_AND_ID.get(int(class_id))\n        if expected_class_name != class_name:\n            logging.debug(\n                f\"The value for class name({class_name}) and class id({class_id}) doesn't match for line {i}. Expected class name: {expected_class_name} for class id: {class_id}.\"\n            )\n\n        if image_id in train_image_ids:\n            whole_output_text_filename = TRAIN_OUTPUT_TXTS.joinpath(f\"{image_id}.txt\")\n            whole_output_image_filename = TRAIN_OUTPUT_IMAGES.joinpath(f\"{image_id}\" + \".png\" if CONVERT_DICOM_TO_PNG else \".dicom\")\n        elif image_id in val_image_ids:\n            whole_output_text_filename = VAL_OUTPUT_TXTS.joinpath(f\"{image_id}.txt\")\n            whole_output_image_filename = VAL_OUTPUT_IMAGES.joinpath(f\"{image_id}\" + \".png\" if CONVERT_DICOM_TO_PNG else \".dicom\")\n        else:\n            whole_output_text_filename = TEST_OUTPUT_TXTS.joinpath(f\"{image_id}.txt\")\n            whole_output_image_filename = TEST_OUTPUT_IMAGES.joinpath(f\"{image_id}\" + \".png\" if CONVERT_DICOM_TO_PNG else \".dicom\")\n        \n        whole_image_filename = INPUT_IMAGES.joinpath(f\"{image_id}\" + \".png\" if CONVERT_DICOM_TO_PNG else \".dicom\")\n        \n        if class_name != \"No finding\": # no need to create a lables file if there is no findings\n            # should produce error if image is not found\n            if os.path.exists(whole_image_filename):\n                image_arr = cv2.imread(str(whole_image_filename))\n            else:\n                image_arr = cv2.imread(str(whole_output_image_filename)) # if already moved\n            \n            image_height = image_arr.shape[0]\n            image_width = image_arr.shape[1]\n            \n            # converting to normalized coordinates\n            x_center = (float(x_max) + float(x_min)) / (2.0 * image_width)\n            y_center = (float(y_max) + float(y_min)) / (2.0 * image_height)\n            width = (float(x_max) - float(x_min)) / (1.0 * image_width)\n            height = (float(y_max) - float(y_min)) / (1.0 * image_height)\n            \n            # writing the a labels file\n            with open(whole_output_text_filename, \"a\") as output_text_file:\n                output_text_file.write(f\"{class_id}\\t{x_center}\\t{y_center}\\t{width}\\t{height}\\n\")\n        log_values = True\n        if COPY_IMAGES:\n            if os.path.exists(whole_image_filename):\n                if DELETE_IMAGES_FROM_PREVIOUS_SOURCE:\n                    shutil.move(whole_image_filename, whole_output_image_filename.parent) # moving to the directory\n                else: # copying the image to the output folder\n                    shutil.copy(whole_image_filename, whole_output_image_filename) # copying as specitic file\n            # if the image file doesn't exist then there is error in some place else \n            # as label file for image whose image doesn't exist shouldn't be created in the first place. \n            # Don't delete the labels files afterwards\n        # only log if not skipped\n        if log_values:\n            logging.debug(values)\n\n        if TESTING:\n            if i > n_max:\n                logging.debug(\n                    f\"The mode was set for testing to break after processing {n_max} entries.\"\n                )\n                break\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:41:11.49208Z","iopub.execute_input":"2024-06-01T02:41:11.492631Z","iopub.status.idle":"2024-06-01T02:41:27.474374Z","shell.execute_reply.started":"2024-06-01T02:41:11.492588Z","shell.execute_reply":"2024-06-01T02:41:27.473306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !tree /kaggle/temp/\"VinBigData Chest X-ray Abnormalities Detection - YOLO\"","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.714452Z","iopub.status.idle":"2024-06-01T02:39:00.71488Z","shell.execute_reply.started":"2024-06-01T02:39:00.714675Z","shell.execute_reply":"2024-06-01T02:39:00.714695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now the folder /kaggle/working/train/labels contains all the labels","metadata":{}},{"cell_type":"code","source":"!du -sh \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO\"","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:41:35.111457Z","iopub.execute_input":"2024-06-01T02:41:35.111832Z","iopub.status.idle":"2024-06-01T02:41:36.150827Z","shell.execute_reply.started":"2024-06-01T02:41:35.111802Z","shell.execute_reply":"2024-06-01T02:41:36.149303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Uploading the dataset as new dataset","metadata":{}},{"cell_type":"code","source":"!mkdir -p ~/.kaggle/\n!echo '{\"username\":\"bihalneupane\",\"key\":\"52d43308ca76e4d785dd4ce2f25b25d5\"}' > ~/.kaggle/kaggle.json","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:41:39.706909Z","iopub.execute_input":"2024-06-01T02:41:39.70737Z","iopub.status.idle":"2024-06-01T02:41:41.764864Z","shell.execute_reply.started":"2024-06-01T02:41:39.707307Z","shell.execute_reply":"2024-06-01T02:41:41.763056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count the number of images\n!find \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO/\" -type f -name '*.png' -printf x | wc -c","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.721186Z","iopub.status.idle":"2024-06-01T02:39:00.721669Z","shell.execute_reply.started":"2024-06-01T02:39:00.721456Z","shell.execute_reply":"2024-06-01T02:39:00.721476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nmetadata = {\n    \"title\": \"VinBigData Chest X-ray - YOLO\",\n    \"id\": \"bihalneupane/VinBigData-Chest-X-ray-YOLO\",\n    \"licenses\": [\n        {\n            \"name\": \"CC0-1.0\"\n        }\n    ]\n}\n\nwith open('/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO/dataset-metadata.json', 'w') as f:\n    json.dump(metadata, f)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.722817Z","iopub.status.idle":"2024-06-01T02:39:00.72321Z","shell.execute_reply.started":"2024-06-01T02:39:00.723025Z","shell.execute_reply":"2024-06-01T02:39:00.723042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !kaggle datasets create -p \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO\" --dir-mode zip","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.724764Z","iopub.status.idle":"2024-06-01T02:39:00.725157Z","shell.execute_reply.started":"2024-06-01T02:39:00.724974Z","shell.execute_reply":"2024-06-01T02:39:00.724992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Appending on pre-exisiting dataset","metadata":{}},{"cell_type":"code","source":"!kaggle datasets metadata bihalneupane/VinBigData-Chest-X-ray-YOLO -p \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO/\"","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.727049Z","iopub.status.idle":"2024-06-01T02:39:00.727454Z","shell.execute_reply.started":"2024-06-01T02:39:00.727245Z","shell.execute_reply":"2024-06-01T02:39:00.727261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO/train\"","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.72869Z","iopub.status.idle":"2024-06-01T02:39:00.729061Z","shell.execute_reply.started":"2024-06-01T02:39:00.728891Z","shell.execute_reply":"2024-06-01T02:39:00.728906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if UPLOAD_TO_KAGGLE:\n    !kaggle datasets version -p \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO/\" -m \"Fix the normalization by calculating height and width of all the images.\" --dir-mode zip","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:39:00.731116Z","iopub.status.idle":"2024-06-01T02:39:00.731512Z","shell.execute_reply.started":"2024-06-01T02:39:00.731302Z","shell.execute_reply":"2024-06-01T02:39:00.731317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Performing Various Checks","metadata":{}},{"cell_type":"code","source":"TRAIN_INPUT_IMAGES = Path(OUTPUT_DIR).joinpath(\"train\", \"images\")\nTRAIN_INPUT_LABELS = Path(OUTPUT_DIR).joinpath(\"train\", \"labels\")\n\nimages_path = glob.glob(str(TRAIN_INPUT_IMAGES) + \"/*\")[:5] # input images path\n\nimg_array = [cv2.imread(image_path) for image_path in images_path]\nplot_imgs(img_array)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:43:37.835964Z","iopub.execute_input":"2024-06-01T02:43:37.836381Z","iopub.status.idle":"2024-06-01T02:43:39.834761Z","shell.execute_reply.started":"2024-06-01T02:43:37.83635Z","shell.execute_reply":"2024-06-01T02:43:39.833567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\ndef plot_image_with_bounding_box(image, bounding_boxes, class_dict):\n    fig, ax = plt.subplots()\n    ax.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    \n    for box in bounding_boxes:\n        class_id, x, y, width, height = map(float, box.split())\n        image_width, image_height = image.shape[1], image.shape[0]\n        x1 = int((x - width / 2) * image_width)\n        y1 = int((y - height / 2) * image_height)\n        x2 = int((x + width / 2) * image_width)\n        y2 = int((y + height / 2) * image_height)\n        \n        # Choose random color for bounding box\n        color = [random.random() for _ in range(3)]\n        \n        rect = plt.Rectangle((x1, y1), x2 - x1, y2 - y1, linewidth=2, edgecolor=color, facecolor='none')\n        ax.add_patch(rect)\n        \n        # Add label text using class label from dictionary\n        label_text = class_dict[int(class_id)]\n        ax.text(x1, y1, label_text, color='white', verticalalignment='top', bbox={'color': color, 'pad': 0})\n    \n    plt.show()\n\ndef plot_images_with_bounding_box_from_path(image_path, bounding_box_path, class_dict):\n    # Read image\n    image = cv2.imread(image_path)\n    \n    # Read bounding boxes\n    try:\n        with open(bounding_box_path, 'r') as file:\n            bounding_boxes = file.readlines()\n\n        # Plot image with bounding boxes\n        plot_image_with_bounding_box(image, bounding_boxes, class_dict)\n    except: # For No Findings\n        plot_imgs([image])\n        \n    \nimage_files = glob.glob(str(TRAIN_INPUT_IMAGES) + \"/*\") \nlabels_files = glob.glob(str(TRAIN_INPUT_LABELS) + \"/*\")\n\nimg_index = 28\nimage_file= image_files[img_index]\nlabels_file = TRAIN_INPUT_LABELS.joinpath(os.path.splitext(os.path.basename(image_file))[0] + \".txt\")\n\nplot_images_with_bounding_box_from_path(image_file, labels_file, CLASS_NAME_AND_ID)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T02:53:29.105903Z","iopub.execute_input":"2024-06-01T02:53:29.106305Z","iopub.status.idle":"2024-06-01T02:53:29.473756Z","shell.execute_reply.started":"2024-06-01T02:53:29.106271Z","shell.execute_reply":"2024-06-01T02:53:29.472409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tree \"/kaggle/temp/VinBigData Chest X-ray Abnormalities Detection - YOLO/\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}