{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":1930191,"datasetId":1151366,"databundleVersionId":1968785},{"sourceType":"datasetVersion","sourceId":951996,"datasetId":516716,"databundleVersionId":979875},{"sourceType":"datasetVersion","sourceId":1799839,"datasetId":1069682,"databundleVersionId":1837296},{"sourceType":"datasetVersion","sourceId":1800066,"datasetId":1069809,"databundleVersionId":1837523},{"sourceType":"datasetVersion","sourceId":1800067,"datasetId":1069810,"databundleVersionId":1837524},{"sourceType":"datasetVersion","sourceId":1800825,"datasetId":1070245,"databundleVersionId":1838284},{"sourceType":"datasetVersion","sourceId":1800777,"datasetId":1069787,"databundleVersionId":1838236},{"sourceType":"datasetVersion","sourceId":1800778,"datasetId":1070222,"databundleVersionId":1838237},{"sourceType":"kernelVersion","sourceId":52422980}],"dockerImageVersionId":30034,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Version\n\n* `v12`: Fold4\n* `v11`: Fold3\n* `v10`: Fold2\n* `v08`: Fold1\n* `v07`: Fold0\n","metadata":{}},{"cell_type":"markdown","source":"# [Training Notebook](https://www.kaggle.com/awsaf49/vinbigdata-cxr-ad-yolov5-14-class)\n* Select `GPU` as the **Accelerator**","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:49.377264Z","iopub.execute_input":"2024-06-11T06:02:49.377649Z","iopub.status.idle":"2024-06-11T06:02:50.241329Z","shell.execute_reply.started":"2024-06-11T06:02:49.377608Z","shell.execute_reply":"2024-06-11T06:02:50.240439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"openi_df=pd.read_csv('/kaggle/input/chest-xrays-indiana-university/indiana_projections.csv')\n# Filter to only include 'Frontal' projections\nfrontal_df = openi_df[openi_df['projection'] == 'Frontal']\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:50.243273Z","iopub.execute_input":"2024-06-11T06:02:50.243562Z","iopub.status.idle":"2024-06-11T06:02:50.274483Z","shell.execute_reply.started":"2024-06-11T06:02:50.243532Z","shell.execute_reply":"2024-06-11T06:02:50.273582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"openi_df2=pd.read_csv('/kaggle/input/chest-xrays-indiana-university/indiana_reports.csv')\nlen(openi_df2)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:50.276018Z","iopub.execute_input":"2024-06-11T06:02:50.27635Z","iopub.status.idle":"2024-06-11T06:02:50.325451Z","shell.execute_reply.started":"2024-06-11T06:02:50.276313Z","shell.execute_reply":"2024-06-11T06:02:50.324472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frontal_df=pd.merge(frontal_df, openi_df2, on=\"uid\", how=\"inner\")","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:50.326559Z","iopub.execute_input":"2024-06-11T06:02:50.326842Z","iopub.status.idle":"2024-06-11T06:02:50.341875Z","shell.execute_reply.started":"2024-06-11T06:02:50.326815Z","shell.execute_reply":"2024-06-11T06:02:50.340873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"openi_folder='/kaggle/input/chest-xrays-indiana-university/images/images_normalized'\nworking_dir='/kaggle/working/'","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:50.345278Z","iopub.execute_input":"2024-06-11T06:02:50.345558Z","iopub.status.idle":"2024-06-11T06:02:50.349381Z","shell.execute_reply.started":"2024-06-11T06:02:50.345531Z","shell.execute_reply":"2024-06-11T06:02:50.348399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(frontal_df[:5])","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:50.35241Z","iopub.execute_input":"2024-06-11T06:02:50.352834Z","iopub.status.idle":"2024-06-11T06:02:50.366903Z","shell.execute_reply.started":"2024-06-11T06:02:50.352791Z","shell.execute_reply":"2024-06-11T06:02:50.366002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nfrom tqdm import tqdm\nfrom PIL import Image\nimport cv2\n\n# Define paths\nopeni_folder = '/kaggle/input/chest-xrays-indiana-university/images/images_normalized'\nworking_dir = '/kaggle/working/'\ntarget_dir = os.path.join(working_dir, 'copied_images')\n\n# Create the target directory if it doesn't exist\nos.makedirs(target_dir, exist_ok=True)\n\n# Function to resize and save images\ndef resize_and_save_image(source_path, target_path, size=(512, 512)):\n    with Image.open(source_path) as img:\n        img_resized = img.resize(size)\n        img_resized.save(target_path)\n\n# Iterate over DataFrame and copy each file with resizing and tqdm progress bar\nfor idx, row in tqdm(frontal_df.iterrows(), total=frontal_df.shape[0], desc=\"Copying and resizing images\"):\n    source_path = os.path.join(openi_folder, row['filename'])\n    target_path = os.path.join(target_dir, row['filename'])\n    resize_and_save_image(source_path, target_path)\n\nprint(f\"Copied and resized {len(frontal_df)} images to {target_dir}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:02:50.368301Z","iopub.execute_input":"2024-06-11T06:02:50.36865Z","iopub.status.idle":"2024-06-11T06:13:21.369166Z","shell.execute_reply.started":"2024-06-11T06:02:50.368617Z","shell.execute_reply":"2024-06-11T06:13:21.368074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:13:21.370967Z","iopub.execute_input":"2024-06-11T06:13:21.371354Z","iopub.status.idle":"2024-06-11T06:13:21.376002Z","shell.execute_reply.started":"2024-06-11T06:13:21.371315Z","shell.execute_reply":"2024-06-11T06:13:21.374957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"I=cv2.imread('/kaggle/working/copied_images/2363_IM-0926-1001.dcm.png')\nprint(I.shape)\nplt.imshow(I)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:13:21.377302Z","iopub.execute_input":"2024-06-11T06:13:21.377659Z","iopub.status.idle":"2024-06-11T06:13:21.64455Z","shell.execute_reply.started":"2024-06-11T06:13:21.377622Z","shell.execute_reply":"2024-06-11T06:13:21.643433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"I=cv2.imread('/kaggle/input/vinbigdata-512-image-dataset/vinbigdata/test/002a34c58c5b758217ed1f584ccbcfe9.png')\nprint(I.shape)\nplt.imshow(I)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:13:21.645936Z","iopub.execute_input":"2024-06-11T06:13:21.646233Z","iopub.status.idle":"2024-06-11T06:13:21.874992Z","shell.execute_reply.started":"2024-06-11T06:13:21.646202Z","shell.execute_reply":"2024-06-11T06:13:21.873641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Directory","metadata":{}},{"cell_type":"code","source":"dim = 512 #1024, 256, 'original'\ntest_dir = f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/test'\ntest_dir='/kaggle/working/copied_images'\nweights_dir = '/kaggle/input/vinbigdata-cxr-ad-yolov5-14-class-train/yolov5/runs/train/exp/weights/best.pt'\n#weights_dir = '/kaggle/input/temp-wbf/best.pt'","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:13:21.876632Z","iopub.execute_input":"2024-06-11T06:13:21.877014Z","iopub.status.idle":"2024-06-11T06:13:21.882564Z","shell.execute_reply.started":"2024-06-11T06:13:21.876976Z","shell.execute_reply":"2024-06-11T06:13:21.881273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:13:21.884099Z","iopub.execute_input":"2024-06-11T06:13:21.884525Z","iopub.status.idle":"2024-06-11T06:13:21.914137Z","shell.execute_reply.started":"2024-06-11T06:13:21.884482Z","shell.execute_reply":"2024-06-11T06:13:21.913175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom PIL import Image\n\n# Define the path to the training folder\ntraining_folder = '/kaggle/input/chest-xrays-indiana-university/images/images_normalized'\ncopied_images_folder = '/kaggle/working/copied_images'\n\n# Get the list of filenames from the copied_images directory\ncopied_image_files = os.listdir(copied_images_folder)\n\n# Initialize lists to store file names (without extensions), widths, and heights\nfile_names = []\nimage_widths = []\nimage_heights = []\n\n# Iterate over the files in the copied_images directory\nfor filename in copied_image_files:\n    if filename.endswith(\".png\"):  # Adjust the extension as needed\n        # Get the file name without extension\n        file_name_no_ext = os.path.splitext(filename)[0]\n        # Remove '.dcm' from the file name if it exists\n        file_name_no_ext = file_name_no_ext.replace('.dcm', '')\n        \n        # Open the image to get its dimensions\n        with Image.open(os.path.join(training_folder, filename)) as img:\n            width, height = img.size\n        \n        # Append the information to the lists\n        file_names.append(file_name_no_ext)\n        image_widths.append(width)\n        image_heights.append(height)\n\n# Create a DataFrame with the collected data\ntest_df = pd.DataFrame({\n    'image_id': file_names,\n    'width': image_widths,\n    'height': image_heights\n})\n\n# Display the first few rows of the DataFrame\nprint(test_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:13:21.915346Z","iopub.execute_input":"2024-06-11T06:13:21.915748Z","iopub.status.idle":"2024-06-11T06:13:24.917863Z","shell.execute_reply.started":"2024-06-11T06:13:21.915698Z","shell.execute_reply":"2024-06-11T06:13:24.916536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# YOLOv5 Stuff","metadata":{}},{"cell_type":"code","source":"shutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\n# shutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', './yolov5')\n\nos.chdir('/kaggle/working/yolov5') # install dependencies\n\nimport torch\nfrom IPython.display import Image, clear_output  # to display images\n\nclear_output()\nprint('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2024-06-11T06:13:24.920018Z","iopub.execute_input":"2024-06-11T06:13:24.920689Z","iopub.status.idle":"2024-06-11T06:13:26.848125Z","shell.execute_reply.started":"2024-06-11T06:13:24.920629Z","shell.execute_reply":"2024-06-11T06:13:26.846991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"!python detect.py --weights $weights_dir\\\n--img 640\\\n--conf 0.15\\\n--iou 0.4\\\n--source $test_dir\\\n--save-txt --save-conf --exist-ok","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-06-11T06:13:26.849918Z","iopub.execute_input":"2024-06-11T06:13:26.850362Z","iopub.status.idle":"2024-06-11T06:18:00.420873Z","shell.execute_reply.started":"2024-06-11T06:13:26.850317Z","shell.execute_reply":"2024-06-11T06:18:00.419621Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport numpy as np\nimport random\nimport cv2\nfrom glob import glob\nfrom tqdm import tqdm\n\nfiles = glob('runs/detect/exp/*png')\nfor _ in range(3):\n    row = 2\n    col = 2\n    grid_files = random.sample(files, row*col)\n    images     = []\n    for image_path in tqdm(grid_files):\n        img          = cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n        images.append(img)\n\n    fig = plt.figure(figsize=(col*5, row*5))\n    grid = ImageGrid(fig, 111,  # similar to subplot(111)\n                     nrows_ncols=(col, row),  # creates 2x2 grid of axes\n                     axes_pad=0.05,  # pad between axes in inch.\n                     )\n\n    for ax, im in zip(grid, images):\n        # Iterating over the grid returns the Axes.\n        ax.imshow(im)\n        ax.set_xticks([])\n        ax.set_yticks([])\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-11T06:18:00.422793Z","iopub.execute_input":"2024-06-11T06:18:00.423106Z","iopub.status.idle":"2024-06-11T06:18:02.134773Z","shell.execute_reply.started":"2024-06-11T06:18:00.423073Z","shell.execute_reply":"2024-06-11T06:18:02.133758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Submission","metadata":{}},{"cell_type":"code","source":"def yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n    \n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n    \n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n    \n    return bboxes","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-11T06:18:02.136641Z","iopub.execute_input":"2024-06-11T06:18:02.137201Z","iopub.status.idle":"2024-06-11T06:18:02.150555Z","shell.execute_reply.started":"2024-06-11T06:18:02.13715Z","shell.execute_reply":"2024-06-11T06:18:02.149056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = []\nPredictionStrings = []\n\nfor file_path in tqdm(glob('runs/detect/exp/labels/*txt')):\n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.image_id==image_id,['width', 'height']].values[0]\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 1).astype(str))\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-06-11T06:18:02.152161Z","iopub.execute_input":"2024-06-11T06:18:02.152546Z","iopub.status.idle":"2024-06-11T06:18:08.809589Z","shell.execute_reply.started":"2024-06-11T06:18:02.152491Z","shell.execute_reply":"2024-06-11T06:18:08.808609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'image_id':image_ids,\n                        'PredictionString':PredictionStrings})\nsub_df = pd.merge(test_df, pred_df, on = 'image_id', how = 'left').fillna(\"14 1 0 0 1 1\")\nsub_df = sub_df[['image_id', 'PredictionString']]\nsub_df.to_csv('/kaggle/working/submission_wbf.csv',index = False)\nsub_df.tail(30)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:08.810983Z","iopub.execute_input":"2024-06-11T06:18:08.811313Z","iopub.status.idle":"2024-06-11T06:18:09.377618Z","shell.execute_reply.started":"2024-06-11T06:18:08.811282Z","shell.execute_reply":"2024-06-11T06:18:09.376742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sub_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.378807Z","iopub.execute_input":"2024-06-11T06:18:09.37912Z","iopub.status.idle":"2024-06-11T06:18:09.3855Z","shell.execute_reply.started":"2024-06-11T06:18:09.379088Z","shell.execute_reply":"2024-06-11T06:18:09.384387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shutil.rmtree('/kaggle/working/yolov5')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-11T06:18:09.386885Z","iopub.execute_input":"2024-06-11T06:18:09.387256Z","iopub.status.idle":"2024-06-11T06:18:09.394693Z","shell.execute_reply.started":"2024-06-11T06:18:09.387221Z","shell.execute_reply":"2024-06-11T06:18:09.393804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0 - Aortic enlargement\n1 - Atelectasis\n2 - Calcification\n3 - Cardiomegaly\n4 - Consolidation\n5 - ILD\n6 - Infiltration\n7 - Lung Opacity\n8 - Nodule/Mass\n9 - Other lesion\n10 - Pleural effusion\n11 - Pleural thickening\n12 - Pneumothorax\n13 - Pulmonary fibrosis\n14-  No Finding","metadata":{}},{"cell_type":"code","source":"images_folder='/kaggle/working/copied_images'\nimage_extension='.dcm.png'\nprint(sub_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.395971Z","iopub.execute_input":"2024-06-11T06:18:09.396286Z","iopub.status.idle":"2024-06-11T06:18:09.409411Z","shell.execute_reply.started":"2024-06-11T06:18:09.396252Z","shell.execute_reply":"2024-06-11T06:18:09.408492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Define the pathology list\npathologies = [\n    'Aortic enlargement', 'Atelectasis', 'Calcification', 'Cardiomegaly', 'Consolidation', \n    'ILD', 'Infiltration', 'Lung Opacity', 'Nodule/Mass', 'Other lesion', \n    'Pleural effusion', 'Pleural thickening', 'Pneumothorax', 'Pulmonary fibrosis', 'No Finding'\n]\n\n# Initialize a counter dictionary for pathologies\npathology_count = {pathology: 0 for pathology in pathologies}\n\n# Function to parse PredictionString\ndef parse_prediction_string(pred_str):\n    parts = pred_str.split()\n    i = 0\n    labels = []\n    while i < len(parts):\n        label = int(parts[i])\n        if label in range(len(pathologies)):\n            labels.append(label)\n        if label == 14:\n            i += 2  # Skip the confidence score\n        else:\n            i += 6  # Skip confidence score and bounding box coordinates\n    return labels\n\n# Count occurrences of each pathology\nfor pred_str in sub_df['PredictionString']:\n    labels = parse_prediction_string(pred_str)\n    for label in labels:\n        pathology_count[pathologies[label]] += 1\n\n# Create histogram\nplt.figure(figsize=(10, 6))\nplt.barh(list(pathology_count.keys()), list(pathology_count.values()))\nplt.xlabel('Frequency')\nplt.ylabel('Pathology')\nplt.title('Histogram of Pathologies')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.410809Z","iopub.execute_input":"2024-06-11T06:18:09.411137Z","iopub.status.idle":"2024-06-11T06:18:09.655003Z","shell.execute_reply.started":"2024-06-11T06:18:09.411105Z","shell.execute_reply":"2024-06-11T06:18:09.653953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_folder = os.path.join(working_dir, 'annotated_images')\n\n# Create the target directory if it doesn't exist\nos.makedirs(target_folder, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.656722Z","iopub.execute_input":"2024-06-11T06:18:09.657212Z","iopub.status.idle":"2024-06-11T06:18:09.662568Z","shell.execute_reply.started":"2024-06-11T06:18:09.657165Z","shell.execute_reply":"2024-06-11T06:18:09.661496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/working/annotated_images')","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.663963Z","iopub.execute_input":"2024-06-11T06:18:09.664336Z","iopub.status.idle":"2024-06-11T06:18:09.673411Z","shell.execute_reply.started":"2024-06-11T06:18:09.664288Z","shell.execute_reply":"2024-06-11T06:18:09.672572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.674672Z","iopub.execute_input":"2024-06-11T06:18:09.674984Z","iopub.status.idle":"2024-06-11T06:18:09.688509Z","shell.execute_reply.started":"2024-06-11T06:18:09.674954Z","shell.execute_reply":"2024-06-11T06:18:09.68741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('/kaggle/working/annotated_images'))","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:18:09.689946Z","iopub.execute_input":"2024-06-11T06:18:09.690347Z","iopub.status.idle":"2024-06-11T06:18:09.700437Z","shell.execute_reply.started":"2024-06-11T06:18:09.690301Z","shell.execute_reply":"2024-06-11T06:18:09.699454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sub_df.iloc[0:1300])","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:19:11.594488Z","iopub.execute_input":"2024-06-11T06:19:11.594894Z","iopub.status.idle":"2024-06-11T06:19:11.600648Z","shell.execute_reply.started":"2024-06-11T06:19:11.59486Z","shell.execute_reply":"2024-06-11T06:19:11.599907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom PIL import Image\nfrom tqdm import tqdm\nimport torch  # Import PyTorch for GPU memory management\n\n# Ensure sub_df is defined\ndf = sub_df.iloc[1300:2600]  # Replace with your actual DataFrame\n\n# Pathologies and corresponding colors\npathologies = [\n    'Aortic enlargement', 'Atelectasis', 'Calcification', 'Cardiomegaly', 'Consolidation', \n    'ILD', 'Infiltration', 'Lung Opacity', 'Nodule/Mass', 'Other lesion', \n    'Pleural effusion', 'Pleural thickening', 'Pneumothorax', 'Pulmonary fibrosis', 'No Finding'\n]\ncolors = ['b', 'g', 'r', 'c', 'm', 'y', 'k', 'purple', 'orange', 'pink', 'brown', 'gray', 'olive', 'cyan', 'lime']\n\n# Image folder and extension\nimages_folder = '/kaggle/input/chest-xrays-indiana-university/images/images_normalized'\ntarget_folder = '/kaggle/working/annotated_images'\nimage_extension = '.dcm.png'\n\n# Function to parse PredictionString\ndef parse_prediction_string(pred_str):\n    parts = pred_str.split()\n    i = 0\n    annotations = []\n    while i < len(parts):\n        label = int(parts[i])\n        confidence = float(parts[i + 1])\n        if label not in [0, 11]:  # Skip Aortic enlargement and Pleural thickening\n            if label == 14:\n                # No Finding doesn't have bounding box coordinates\n                annotations.append((label, confidence, None))\n                i += 2\n            else:\n                # Other pathologies have bounding box coordinates\n                x1, y1, x2, y2 = map(int, parts[i + 2:i + 6])\n                annotations.append((label, confidence, (x1, y1, x2, y2)))\n                i += 6\n        else:\n            i += 6  # Skip Aortic enlargement and Pleural thickening entries\n    return annotations\n\n# Ensure target folder exists\nos.makedirs(target_folder, exist_ok=True)\n\n# Batch size to control memory usage\nbatch_size = 150\n\n# Annotate images in batches\ntotal_batches = (len(df) + batch_size - 1) // batch_size  # Calculate total number of batches\nwith tqdm(total=len(df), desc=\"Overall Progress\") as pbar:\n    for start_index in range(0, len(df), batch_size):\n        end_index = min(start_index + batch_size, len(df))\n        batch_df = df.iloc[start_index:end_index]\n\n        for index, row in batch_df.iterrows():\n            image_id = row['image_id']\n            pred_str = row['PredictionString']\n            annotations = parse_prediction_string(pred_str)\n\n            image_path = os.path.join(images_folder, image_id + image_extension)\n\n            if not os.path.exists(image_path):\n                print(f\"File not found: {image_path}\")\n                continue\n\n            image = Image.open(image_path)\n\n            original_size = image.size\n            fig, ax = plt.subplots(figsize=(original_size[0] / 100, original_size[1] / 100), dpi=100)\n            ax.imshow(image, cmap='gray')\n\n            # Track the y-coordinates of the last annotations\n            used_positions = []\n\n            for label, confidence, bbox in annotations:\n                if bbox is not None:\n                    x1, y1, x2, y2 = bbox\n                    width = x2 - x1\n                    height = y2 - y1\n                    rect = patches.Rectangle((x1, y1), width, height, linewidth=2, edgecolor=colors[label], facecolor='none')\n                    ax.add_patch(rect)\n\n                    # Calculate position for text box to avoid overlapping\n                    text_y = y1 - 10\n                    # Ensure that the text does not overlap by adjusting the position\n                    while any(abs(text_y - used_y) < 20 for used_y in used_positions):\n                        text_y -= 20\n                    used_positions.append(text_y)\n\n                    ax.text(x1, text_y, f'{pathologies[label]}: {confidence:.2f}', color=colors[label], fontsize=12, bbox=dict(facecolor='white', alpha=0.5))\n\n            ax.axis('off')\n            save_path = os.path.join(target_folder, f'annotated_{image_id}.png')\n            fig.savefig(save_path, bbox_inches='tight', pad_inches=0, dpi=100)\n            plt.close(fig)  # Close the current plot to avoid accumulation\n            \n            # Update the progress bar\n            pbar.update(1)\n\n        # Clear GPU memory\n        torch.cuda.empty_cache()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T06:27:36.514979Z","iopub.execute_input":"2024-06-11T06:27:36.515334Z","iopub.status.idle":"2024-06-11T07:02:17.739836Z","shell.execute_reply.started":"2024-06-11T06:27:36.515304Z","shell.execute_reply":"2024-06-11T07:02:17.738991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('/kaggle/working/annotated_images'))","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:03:03.363764Z","iopub.execute_input":"2024-06-11T07:03:03.364207Z","iopub.status.idle":"2024-06-11T07:03:03.371099Z","shell.execute_reply.started":"2024-06-11T07:03:03.364167Z","shell.execute_reply":"2024-06-11T07:03:03.370332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:03:07.716414Z","iopub.execute_input":"2024-06-11T07:03:07.716878Z","iopub.status.idle":"2024-06-11T07:03:07.721084Z","shell.execute_reply.started":"2024-06-11T07:03:07.716842Z","shell.execute_reply":"2024-06-11T07:03:07.720016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=random.choice(os.listdir(target_folder))\npath","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:04:50.609458Z","iopub.execute_input":"2024-06-11T07:04:50.609931Z","iopub.status.idle":"2024-06-11T07:04:50.617844Z","shell.execute_reply.started":"2024-06-11T07:04:50.60987Z","shell.execute_reply":"2024-06-11T07:04:50.616663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=random.choice(os.listdir(target_folder))\nI=Image.open(os.path.join(target_folder,path))\noriginal_size=I.size\nprint(original_size)\nfig, ax = plt.subplots(figsize=(original_size[0] / 100, original_size[1] / 100), dpi=100)\nax.imshow(I,cmap='gray')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:09:35.832178Z","iopub.execute_input":"2024-06-11T07:09:35.832572Z","iopub.status.idle":"2024-06-11T07:09:36.926915Z","shell.execute_reply.started":"2024-06-11T07:09:35.832537Z","shell.execute_reply":"2024-06-11T07:09:36.925979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=random.choice(os.listdir(target_folder))\nI=Image.open(os.path.join(target_folder,path))\noriginal_size=I.size\nprint(original_size)\nfig, ax = plt.subplots(figsize=(original_size[0] / 100, original_size[1] / 100), dpi=100)\nax.imshow(I,cmap='gray')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:09:51.30083Z","iopub.execute_input":"2024-06-11T07:09:51.301236Z","iopub.status.idle":"2024-06-11T07:09:52.432455Z","shell.execute_reply.started":"2024-06-11T07:09:51.301198Z","shell.execute_reply":"2024-06-11T07:09:52.429059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=random.choice(os.listdir(target_folder))\nI=Image.open(os.path.join(target_folder,path))\noriginal_size=I.size\nprint(original_size)\nfig, ax = plt.subplots(figsize=(original_size[0] / 100, original_size[1] / 100), dpi=100)\nax.imshow(I,cmap='gray')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:09:58.491324Z","iopub.execute_input":"2024-06-11T07:09:58.491675Z","iopub.status.idle":"2024-06-11T07:09:59.76291Z","shell.execute_reply.started":"2024-06-11T07:09:58.491641Z","shell.execute_reply":"2024-06-11T07:09:59.761543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=random.choice(os.listdir(target_folder))\nI=Image.open(os.path.join(target_folder,path))\noriginal_size=I.size\nprint(original_size)\nfig, ax = plt.subplots(figsize=(original_size[0] / 100, original_size[1] / 100), dpi=100)\nax.imshow(I,cmap='gray')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:10:12.146354Z","iopub.execute_input":"2024-06-11T07:10:12.146823Z","iopub.status.idle":"2024-06-11T07:10:13.399847Z","shell.execute_reply.started":"2024-06-11T07:10:12.14676Z","shell.execute_reply":"2024-06-11T07:10:13.398844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=random.choice(os.listdir(target_folder))\nI=Image.open(os.path.join(target_folder,path))\noriginal_size=I.size\nprint(original_size)\nfig, ax = plt.subplots(figsize=(original_size[0] / 100, original_size[1] / 100), dpi=100)\nax.imshow(I,cmap='gray')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:10:29.326621Z","iopub.execute_input":"2024-06-11T07:10:29.327072Z","iopub.status.idle":"2024-06-11T07:10:30.712538Z","shell.execute_reply.started":"2024-06-11T07:10:29.327033Z","shell.execute_reply":"2024-06-11T07:10:30.711193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clear output folder\nimport os\n\ndef remove_folder_contents(folder):\n    for the_file in os.listdir(folder):\n        file_path = os.path.join(folder, the_file)\n        try:\n            if os.path.isfile(file_path):\n                os.unlink(file_path)\n            elif os.path.isdir(file_path):\n                remove_folder_contents(file_path)\n                os.rmdir(file_path)\n        except Exception as e:\n            print(e)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:10:59.998135Z","iopub.execute_input":"2024-06-11T07:10:59.998551Z","iopub.status.idle":"2024-06-11T07:11:00.006584Z","shell.execute_reply.started":"2024-06-11T07:10:59.998518Z","shell.execute_reply":"2024-06-11T07:11:00.005468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = '/kaggle/working/copied_images'\nremove_folder_contents(folder_path)\nos.rmdir(folder_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:11:04.961034Z","iopub.execute_input":"2024-06-11T07:11:04.961412Z","iopub.status.idle":"2024-06-11T07:11:05.122298Z","shell.execute_reply.started":"2024-06-11T07:11:04.96138Z","shell.execute_reply":"2024-06-11T07:11:05.121512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = '/kaggle/working/yolov5'\nremove_folder_contents(folder_path)\nos.rmdir(folder_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:11:09.29431Z","iopub.execute_input":"2024-06-11T07:11:09.294657Z","iopub.status.idle":"2024-06-11T07:11:09.611633Z","shell.execute_reply.started":"2024-06-11T07:11:09.294626Z","shell.execute_reply":"2024-06-11T07:11:09.610772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n\ndef compress_folder_to_zip(folder_path, output_zip_path):\n    with zipfile.ZipFile(output_zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, _, files in os.walk(folder_path):\n            for file in files:\n                if file.endswith('.png'):  # Add only .png files to the zip\n                    file_path = os.path.join(root, file)\n                    arcname = os.path.relpath(file_path, start=folder_path)\n                    zipf.write(file_path, arcname=arcname)\n    print(f'All files in {folder_path} have been compressed into {output_zip_path}')\n\n# Define folder path and output zip path\nfolder_path = '/kaggle/working/annotated_images'\noutput_zip_path = '/kaggle/working/annotated_images.zip'\n\n# Compress the folder into a zip file\ncompress_folder_to_zip(folder_path, output_zip_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:11:29.841508Z","iopub.execute_input":"2024-06-11T07:11:29.841902Z","iopub.status.idle":"2024-06-11T07:13:11.270354Z","shell.execute_reply.started":"2024-06-11T07:11:29.841867Z","shell.execute_reply":"2024-06-11T07:13:11.269326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Hell')","metadata":{"execution":{"iopub.status.busy":"2024-06-11T07:19:41.849373Z","iopub.execute_input":"2024-06-11T07:19:41.84971Z","iopub.status.idle":"2024-06-11T07:19:41.854475Z","shell.execute_reply.started":"2024-06-11T07:19:41.849681Z","shell.execute_reply":"2024-06-11T07:19:41.85356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}